feat(mnemopi): add embedding variant setting (English / multilingual) (#2476)

This commit is contained in:
metaphorics
2026-06-14 05:55:21 +09:00
parent eac1d21f93
commit dd2e736476
9 changed files with 300 additions and 4 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added the `mnemopi.embeddingVariant` setting (`en` | `multilingual`) selecting a stronger SOTA local embedding model — `en` → `BAAI/bge-base-en-v1.5` (768d), `multilingual` → `intfloat/multilingual-e5-large` (1024d). `mnemopi.embeddingModel` still works as an advanced explicit override (it wins over the variant). Changing the active model wipes and rebuilds stored embeddings on the next start ([#2476](https://github.com/can1357/oh-my-pi/issues/2476))
## [15.12.5] - 2026-06-13
### Changed
@@ -1894,6 +1894,31 @@ export const SETTINGS_SCHEMA = {
condition: "mnemopiActive",
},
},
"mnemopi.embeddingVariant": {
type: "enum",
values: ["en", "multilingual"] as const,
default: "en",
ui: {
tab: "memory",
group: "Mnemopi",
label: "Embedding variant",
description:
"Local embedding model family. en = stronger English model; multilingual = cross-language model. Changing this rebuilds existing memory embeddings on next start.",
options: [
{
value: "en",
label: "English (bge-base-en-v1.5)",
description: "BAAI/bge-base-en-v1.5 (768d), English-only",
},
{
value: "multilingual",
label: "Multilingual (multilingual-e5-large)",
description: "intfloat/multilingual-e5-large (1024d), cross-language recall",
},
],
condition: "mnemopiActive",
},
},
"mnemopi.autoRecall": {
type: "boolean",
default: true,
@@ -1956,7 +1981,8 @@ export const SETTINGS_SCHEMA = {
tab: "memory",
group: "Mnemopi",
label: "Mnemopi Embedding Model",
description: "Optional embedding model override passed to Mnemopi",
description:
"Advanced: explicit embedding model id that overrides the variant. Leave empty to use mnemopi.embeddingVariant.",
condition: "mnemopiActive",
},
},
+9 -1
View File
@@ -48,6 +48,14 @@ export function loadMnemopiConfig(settings: Settings, agentDir: string): Mnemopi
const recallBanks =
scoping === "global" ? scope.recallBanks : extendRecallWithLegacyBanks(scope.recallBanks, dbPath, cwd);
const llmMode = settings.get("mnemopi.llmMode");
const embeddingOverride = settings.get("mnemopi.embeddingModel");
const embeddingVariant = settings.get("mnemopi.embeddingVariant");
// Map the variant explicitly rather than indexing an object with the raw config
// value (which could resolve an inherited property like `__proto__`); any value
// other than the multilingual variant falls back to the English default.
const variantModel =
embeddingVariant === "multilingual" ? "intfloat/multilingual-e5-large" : "BAAI/bge-base-en-v1.5";
const embeddingModel = embeddingOverride?.trim() || variantModel;
return {
dbPath,
baseBank: scope.baseBank,
@@ -69,7 +77,7 @@ export function loadMnemopiConfig(settings: Settings, agentDir: string): Mnemopi
providerOptions: {
noEmbeddings: settings.get("mnemopi.noEmbeddings"),
debug: settings.get("mnemopi.debug"),
embeddingModel: settings.get("mnemopi.embeddingModel"),
embeddingModel,
embeddingApiUrl: settings.get("mnemopi.embeddingApiUrl"),
embeddingApiKey: settings.get("mnemopi.embeddingApiKey"),
llm:
@@ -0,0 +1,36 @@
import { describe, expect, it } from "bun:test";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { loadMnemopiConfig } from "@oh-my-pi/pi-coding-agent/mnemopi/config";
// `mnemopi.embeddingVariant` selects the concrete local embedding model, while an
// explicit `mnemopi.embeddingModel` is an advanced override that wins. Scoping is
// pinned to "global" so the resolver stays pure (no legacy-bank disk probing).
function embeddingModelFor(overrides: Record<string, unknown>): string | undefined {
const settings = Settings.isolated({ "mnemopi.scoping": "global", ...overrides });
return loadMnemopiConfig(settings, "/tmp/mnemopi-embedding-variant-test").providerOptions.embeddingModel;
}
describe("loadMnemopiConfig embedding variant resolution", () => {
it("maps the en variant to BAAI/bge-base-en-v1.5", () => {
expect(embeddingModelFor({ "mnemopi.embeddingVariant": "en" })).toBe("BAAI/bge-base-en-v1.5");
});
it("maps the multilingual variant to intfloat/multilingual-e5-large", () => {
expect(embeddingModelFor({ "mnemopi.embeddingVariant": "multilingual" })).toBe("intfloat/multilingual-e5-large");
});
it("lets an explicit embeddingModel override win over the variant", () => {
expect(
embeddingModelFor({
"mnemopi.embeddingVariant": "multilingual",
"mnemopi.embeddingModel": "openai/text-embedding-3-small",
}),
).toBe("openai/text-embedding-3-small");
});
it("ignores a blank override and falls back to the variant", () => {
expect(embeddingModelFor({ "mnemopi.embeddingVariant": "en", "mnemopi.embeddingModel": " " })).toBe(
"BAAI/bge-base-en-v1.5",
);
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added a wipe-and-rebuild reconcile (`reconcileEmbeddingModel`) that runs when the configured embedding model changes. At store open, if the model stamped on stored `memory_embeddings` rows differs from the active `currentEmbeddingModel()`, the stale embeddings and their binary vectors are dropped and every existing memory is enqueued for background re-embedding (in bounded batches) at the new model/dimension. The destructive wipe is skipped whenever it could not be rebuilt — embeddings disabled via the runtime option or the `MNEMOPI_NO_EMBEDDINGS` env, or an unresolved (empty) active model — so a stale-but-valid corpus is never destroyed without a replacement. Recall degrades gracefully (FTS-only) for memories whose vectors are not yet rebuilt ([#2476](https://github.com/can1357/oh-my-pi/issues/2476))
## [15.12.4] - 2026-06-13
### Fixed
+74 -1
View File
@@ -1,12 +1,14 @@
import type { Database, SQLQueryBindings } from "bun:sqlite";
import { logger } from "@oh-my-pi/pi-utils";
import { transaction } from "../../db";
import { toUtcIso } from "../../util/datetime";
import { generateId } from "../../util/ids";
import { currentEmbeddingModel, embeddingsDisabled } from "../embeddings";
import { EpisodicGraph } from "../episodic-graph";
import { extractFactsSafe } from "../extraction";
import { getMnemopiRuntimeOptions, withMnemopiRuntimeOptions } from "../runtime-options";
import { storeFactStrings } from "./consolidate";
import { scheduleEmbedding, vecAvailable, vecInsert } from "./helpers";
import { type EmbedItem, scheduleEmbedding, vecAvailable, vecInsert } from "./helpers";
import type {
BeamEvent,
BeamMemoryState,
@@ -248,6 +250,77 @@ function rowToDict(row: Row): Row {
return { ...row };
}
/** Re-embedding batch size for a model-change rebuild — bounds each background
* embedding request instead of embedding the whole corpus in one call. */
const EMBED_REBUILD_BATCH = 128;
/**
* Reconcile stored embeddings against the active embedding model at store open.
*
* Every `memory_embeddings` row is stamped with the model that produced it (see
* `runEmbedding` in `helpers.ts`). When the configured embedding model changes,
* its vector dimension changes too, so the previously-stored vectors are no
* longer comparable. On a mismatch we wipe every stored vector — the
* `memory_embeddings` table, the `episodic_memory.binary_vector` column, and the
* sqlite-vec `vec_episodes` index — then enqueue all live memories for
* background re-embedding under the new model via `scheduleEmbedding`.
*
* Runs once per store open; a fresh store (no embeddings) or an already-current
* store is a no-op. The destructive wipe is skipped whenever it could not be
* rebuilt — embeddings disabled via the runtime option OR the
* `MNEMOPI_NO_EMBEDDINGS` env, or an unresolved (empty) active model — so a
* stale-but-valid corpus is never destroyed without a replacement. MUST run
* inside the active runtime-options scope so `currentEmbeddingModel()` /
* `embeddingsDisabled()` reflect the per-instance configuration.
*/
export function reconcileEmbeddingModel(beam: BeamMemoryState): void {
if (embeddingsDisabled()) return;
const active = currentEmbeddingModel().trim();
if (active === "") return;
// Stop at the first row whose stamped model differs from the active one
// (NULL/unstamped counts as a mismatch via `IS NOT`); an empty store or an
// all-current store short-circuits here without gathering the DISTINCT set.
const mismatch = beam.db.query("SELECT 1 FROM memory_embeddings WHERE model IS NOT ? LIMIT 1").get(active);
if (!mismatch) return;
const staleModels = beam.db
.query("SELECT DISTINCT model FROM memory_embeddings WHERE model IS NOT ?")
.all(active) as { model: string | null }[];
const live = beam.db
.query(`
SELECT id AS memoryId, content FROM working_memory WHERE superseded_by IS NULL
UNION ALL
SELECT id AS memoryId, content FROM episodic_memory WHERE superseded_by IS NULL
`)
.all() as EmbedItem[];
transaction(beam.db, () => {
beam.db.prepare("DELETE FROM memory_embeddings").run();
beam.db.prepare("UPDATE episodic_memory SET binary_vector = NULL").run();
if (vecAvailable(beam.db)) {
try {
beam.db.prepare("DELETE FROM vec_episodes").run();
} catch {
// sqlite-vec cleanup is best-effort; rebuild correctness takes precedence.
}
}
});
logger.info("mnemopi: embedding model changed, rebuilding", {
from: staleModels.map(row => row.model ?? "(unstamped)"),
to: active,
count: live.length,
});
// Re-embed in bounded batches so a corpus-wide rebuild never issues one giant
// embedding request; each batch is its own tracked background task.
for (let offset = 0; offset < live.length; offset += EMBED_REBUILD_BATCH) {
scheduleEmbedding(beam, live.slice(offset, offset + EMBED_REBUILD_BATCH));
}
}
export function remember(beam: BeamMemoryState, content: string, options: StoreRememberOptions = {}): string {
const source = options.source ?? "conversation";
const importance = options.importance ?? 0.5;
+1 -1
View File
@@ -99,7 +99,7 @@ function inTestRuntime(): boolean {
return $env.NODE_ENV === "test" || $env.BUN_ENV === "test";
}
function embeddingsDisabled(): boolean {
export function embeddingsDisabled(): boolean {
const active = activeEmbeddingOptions();
if (active?.disabled !== undefined) {
return active.disabled;
+5
View File
@@ -7,6 +7,7 @@ import type { MemoryInput, Metadata } from "../types";
import { AnnotationStore } from "./annotations";
import { BankManager } from "./banks";
import { BeamMemory, initBeam } from "./beam/index";
import { reconcileEmbeddingModel } from "./beam/store";
import type { RecallEnhancedOptions, RecallOptions, RecallResult, SleepResult } from "./beam/types";
import { EpisodicGraph } from "./episodic-graph";
import {
@@ -388,6 +389,10 @@ export class Mnemopi {
}
this.conn = this.beam.db;
this.db = this.beam.db;
// Wipe-and-rebuild stale embeddings when the configured model changed since
// the vectors were written. Runs inside the runtime scope so
// `currentEmbeddingModel()` reflects this instance's configured model.
this.#withRuntimeOptions(() => reconcileEmbeddingModel(this.beam));
}
close(): void {
@@ -0,0 +1,140 @@
/**
* Wipe-and-rebuild reconcile: each `memory_embeddings` row is stamped with the
* model that produced it. When the configured embedding model changes the vector
* dimension changes too, so on store open `reconcileEmbeddingModel` (wired into
* the `Mnemopi` constructor) wipes every stale vector and enqueues all live
* memories for re-embedding under the new model. A matching model is a no-op.
*/
import { Database } from "bun:sqlite";
import { describe, expect, it } from "bun:test";
import "./setup";
import { initBeam } from "@oh-my-pi/pi-mnemopi/core/beam";
import { Mnemopi } from "@oh-my-pi/pi-mnemopi/core/memory";
const OLD_MODEL = "BAAI/bge-small-en-v1.5";
const NEW_MODEL = "intfloat/multilingual-e5-large";
// Deterministic fastembed-shaped provider so the background rebuild actually
// writes rows (and stamps them with the active model) under the test runtime.
function fakeEmbed() {
return async function* embed(texts: readonly string[]) {
yield texts.map(() => [0.1, 0.2, 0.3, 0.4]);
};
}
function seedDb(model: string): { db: Database; ids: string[] } {
const db = new Database(":memory:");
initBeam(db);
const ts = new Date().toISOString();
db.prepare(
"INSERT INTO working_memory (id, content, source, timestamp, session_id) VALUES (?, ?, 'test', ?, 'default')",
).run("wm-1", "alpha working memory", ts);
db.prepare(
"INSERT INTO episodic_memory (id, content, source, timestamp, session_id, binary_vector) VALUES (?, ?, 'test', ?, 'default', ?)",
).run("ep-1", "beta episodic memory", ts, new Uint8Array([1, 2, 3, 4]));
for (const id of ["wm-1", "ep-1"]) {
db.prepare("INSERT INTO memory_embeddings (memory_id, embedding_json, model) VALUES (?, ?, ?)").run(
id,
JSON.stringify([1, 0, 0, 0]),
model,
);
}
return { db, ids: ["wm-1", "ep-1"] };
}
function countEmbeddings(memory: Mnemopi): number {
return (memory.conn.query("SELECT COUNT(*) AS n FROM memory_embeddings").get() as { n: number }).n;
}
describe("reconcileEmbeddingModel on store open", () => {
it("wipes stale embeddings + binary vectors and re-embeds when the model changed", async () => {
const { db, ids } = seedDb(OLD_MODEL);
const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
try {
// Reconcile fired in the constructor: stale vector rows are gone and the
// episodic binary vector was cleared. The async rebuild is enqueued but
// has not yet run.
expect(countEmbeddings(memory)).toBe(0);
const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as {
v: Uint8Array | null;
};
expect(ep.v).toBeNull();
expect(memory.beam.pendingExtractions.size).toBeGreaterThanOrEqual(1);
// The background rebuild repopulates every live memory, stamped with the
// new model.
await memory.flushExtractions();
const rows = memory.conn.query("SELECT memory_id, model FROM memory_embeddings ORDER BY memory_id").all() as {
memory_id: string;
model: string;
}[];
expect(rows.map(row => row.memory_id).sort()).toEqual([...ids].sort());
expect(rows.every(row => row.model === NEW_MODEL)).toBe(true);
} finally {
memory.close();
db.close();
}
});
it("leaves embeddings untouched when the stored model already matches", () => {
const { db } = seedDb(NEW_MODEL);
const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
try {
// No mismatch -> no wipe, no rebuild enqueued.
expect(countEmbeddings(memory)).toBe(2);
expect(memory.beam.pendingExtractions.size).toBe(0);
// The no-op path must preserve the episodic binary vector too (regression
// guard against an unconditional clear that would silently drop vectors).
const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as {
v: Uint8Array | null;
};
expect(ep.v).not.toBeNull();
expect(Array.from(ep.v as Uint8Array)).toEqual([1, 2, 3, 4]);
} finally {
memory.close();
db.close();
}
});
it("does not wipe when embeddings are disabled via the MNEMOPI_NO_EMBEDDINGS env", () => {
const { db } = seedDb(OLD_MODEL);
const previous = process.env.MNEMOPI_NO_EMBEDDINGS;
process.env.MNEMOPI_NO_EMBEDDINGS = "1";
let memory: Mnemopi | undefined;
try {
// The model differs, but with embeddings disabled the rebuild would
// produce nothing — so the stale-but-present vectors must survive.
memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
expect(countEmbeddings(memory)).toBe(2);
expect(memory.beam.pendingExtractions.size).toBe(0);
const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as {
v: Uint8Array | null;
};
expect(ep.v).not.toBeNull();
} finally {
if (previous === undefined) {
delete process.env.MNEMOPI_NO_EMBEDDINGS;
} else {
process.env.MNEMOPI_NO_EMBEDDINGS = previous;
}
memory?.close();
db.close();
}
});
it("does not wipe when the active embedding model is empty", () => {
const { db } = seedDb(OLD_MODEL);
let memory: Mnemopi | undefined;
try {
// An explicit empty model resolves to no embedder; wiping would be
// unrecoverable, so the reconcile must skip.
memory = new Mnemopi({ db, embeddings: { model: "", provider: fakeEmbed() } });
expect(countEmbeddings(memory)).toBe(2);
expect(memory.beam.pendingExtractions.size).toBe(0);
} finally {
memory?.close();
db.close();
}
});
});