feat(mnemopi): add embedding variant setting (English / multilingual) (#2476)
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added the `mnemopi.embeddingVariant` setting (`en` | `multilingual`) selecting a stronger SOTA local embedding model — `en` → `BAAI/bge-base-en-v1.5` (768d), `multilingual` → `intfloat/multilingual-e5-large` (1024d). `mnemopi.embeddingModel` still works as an advanced explicit override (it wins over the variant). Changing the active model wipes and rebuilds stored embeddings on the next start ([#2476](https://github.com/can1357/oh-my-pi/issues/2476))
|
||||
|
||||
## [15.12.5] - 2026-06-13
|
||||
### Changed
|
||||
|
||||
|
||||
@@ -1894,6 +1894,31 @@ export const SETTINGS_SCHEMA = {
|
||||
condition: "mnemopiActive",
|
||||
},
|
||||
},
|
||||
"mnemopi.embeddingVariant": {
|
||||
type: "enum",
|
||||
values: ["en", "multilingual"] as const,
|
||||
default: "en",
|
||||
ui: {
|
||||
tab: "memory",
|
||||
group: "Mnemopi",
|
||||
label: "Embedding variant",
|
||||
description:
|
||||
"Local embedding model family. en = stronger English model; multilingual = cross-language model. Changing this rebuilds existing memory embeddings on next start.",
|
||||
options: [
|
||||
{
|
||||
value: "en",
|
||||
label: "English (bge-base-en-v1.5)",
|
||||
description: "BAAI/bge-base-en-v1.5 (768d), English-only",
|
||||
},
|
||||
{
|
||||
value: "multilingual",
|
||||
label: "Multilingual (multilingual-e5-large)",
|
||||
description: "intfloat/multilingual-e5-large (1024d), cross-language recall",
|
||||
},
|
||||
],
|
||||
condition: "mnemopiActive",
|
||||
},
|
||||
},
|
||||
"mnemopi.autoRecall": {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
@@ -1956,7 +1981,8 @@ export const SETTINGS_SCHEMA = {
|
||||
tab: "memory",
|
||||
group: "Mnemopi",
|
||||
label: "Mnemopi Embedding Model",
|
||||
description: "Optional embedding model override passed to Mnemopi",
|
||||
description:
|
||||
"Advanced: explicit embedding model id that overrides the variant. Leave empty to use mnemopi.embeddingVariant.",
|
||||
condition: "mnemopiActive",
|
||||
},
|
||||
},
|
||||
|
||||
@@ -48,6 +48,14 @@ export function loadMnemopiConfig(settings: Settings, agentDir: string): Mnemopi
|
||||
const recallBanks =
|
||||
scoping === "global" ? scope.recallBanks : extendRecallWithLegacyBanks(scope.recallBanks, dbPath, cwd);
|
||||
const llmMode = settings.get("mnemopi.llmMode");
|
||||
const embeddingOverride = settings.get("mnemopi.embeddingModel");
|
||||
const embeddingVariant = settings.get("mnemopi.embeddingVariant");
|
||||
// Map the variant explicitly rather than indexing an object with the raw config
|
||||
// value (which could resolve an inherited property like `__proto__`); any value
|
||||
// other than the multilingual variant falls back to the English default.
|
||||
const variantModel =
|
||||
embeddingVariant === "multilingual" ? "intfloat/multilingual-e5-large" : "BAAI/bge-base-en-v1.5";
|
||||
const embeddingModel = embeddingOverride?.trim() || variantModel;
|
||||
return {
|
||||
dbPath,
|
||||
baseBank: scope.baseBank,
|
||||
@@ -69,7 +77,7 @@ export function loadMnemopiConfig(settings: Settings, agentDir: string): Mnemopi
|
||||
providerOptions: {
|
||||
noEmbeddings: settings.get("mnemopi.noEmbeddings"),
|
||||
debug: settings.get("mnemopi.debug"),
|
||||
embeddingModel: settings.get("mnemopi.embeddingModel"),
|
||||
embeddingModel,
|
||||
embeddingApiUrl: settings.get("mnemopi.embeddingApiUrl"),
|
||||
embeddingApiKey: settings.get("mnemopi.embeddingApiKey"),
|
||||
llm:
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { loadMnemopiConfig } from "@oh-my-pi/pi-coding-agent/mnemopi/config";
|
||||
|
||||
// `mnemopi.embeddingVariant` selects the concrete local embedding model, while an
|
||||
// explicit `mnemopi.embeddingModel` is an advanced override that wins. Scoping is
|
||||
// pinned to "global" so the resolver stays pure (no legacy-bank disk probing).
|
||||
function embeddingModelFor(overrides: Record<string, unknown>): string | undefined {
|
||||
const settings = Settings.isolated({ "mnemopi.scoping": "global", ...overrides });
|
||||
return loadMnemopiConfig(settings, "/tmp/mnemopi-embedding-variant-test").providerOptions.embeddingModel;
|
||||
}
|
||||
|
||||
describe("loadMnemopiConfig embedding variant resolution", () => {
|
||||
it("maps the en variant to BAAI/bge-base-en-v1.5", () => {
|
||||
expect(embeddingModelFor({ "mnemopi.embeddingVariant": "en" })).toBe("BAAI/bge-base-en-v1.5");
|
||||
});
|
||||
|
||||
it("maps the multilingual variant to intfloat/multilingual-e5-large", () => {
|
||||
expect(embeddingModelFor({ "mnemopi.embeddingVariant": "multilingual" })).toBe("intfloat/multilingual-e5-large");
|
||||
});
|
||||
|
||||
it("lets an explicit embeddingModel override win over the variant", () => {
|
||||
expect(
|
||||
embeddingModelFor({
|
||||
"mnemopi.embeddingVariant": "multilingual",
|
||||
"mnemopi.embeddingModel": "openai/text-embedding-3-small",
|
||||
}),
|
||||
).toBe("openai/text-embedding-3-small");
|
||||
});
|
||||
|
||||
it("ignores a blank override and falls back to the variant", () => {
|
||||
expect(embeddingModelFor({ "mnemopi.embeddingVariant": "en", "mnemopi.embeddingModel": " " })).toBe(
|
||||
"BAAI/bge-base-en-v1.5",
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added a wipe-and-rebuild reconcile (`reconcileEmbeddingModel`) that runs when the configured embedding model changes. At store open, if the model stamped on stored `memory_embeddings` rows differs from the active `currentEmbeddingModel()`, the stale embeddings and their binary vectors are dropped and every existing memory is enqueued for background re-embedding (in bounded batches) at the new model/dimension. The destructive wipe is skipped whenever it could not be rebuilt — embeddings disabled via the runtime option or the `MNEMOPI_NO_EMBEDDINGS` env, or an unresolved (empty) active model — so a stale-but-valid corpus is never destroyed without a replacement. Recall degrades gracefully (FTS-only) for memories whose vectors are not yet rebuilt ([#2476](https://github.com/can1357/oh-my-pi/issues/2476))
|
||||
|
||||
## [15.12.4] - 2026-06-13
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
import type { Database, SQLQueryBindings } from "bun:sqlite";
|
||||
import { logger } from "@oh-my-pi/pi-utils";
|
||||
import { transaction } from "../../db";
|
||||
import { toUtcIso } from "../../util/datetime";
|
||||
import { generateId } from "../../util/ids";
|
||||
import { currentEmbeddingModel, embeddingsDisabled } from "../embeddings";
|
||||
import { EpisodicGraph } from "../episodic-graph";
|
||||
import { extractFactsSafe } from "../extraction";
|
||||
import { getMnemopiRuntimeOptions, withMnemopiRuntimeOptions } from "../runtime-options";
|
||||
import { storeFactStrings } from "./consolidate";
|
||||
import { scheduleEmbedding, vecAvailable, vecInsert } from "./helpers";
|
||||
import { type EmbedItem, scheduleEmbedding, vecAvailable, vecInsert } from "./helpers";
|
||||
import type {
|
||||
BeamEvent,
|
||||
BeamMemoryState,
|
||||
@@ -248,6 +250,77 @@ function rowToDict(row: Row): Row {
|
||||
return { ...row };
|
||||
}
|
||||
|
||||
/** Re-embedding batch size for a model-change rebuild — bounds each background
|
||||
* embedding request instead of embedding the whole corpus in one call. */
|
||||
const EMBED_REBUILD_BATCH = 128;
|
||||
|
||||
/**
|
||||
* Reconcile stored embeddings against the active embedding model at store open.
|
||||
*
|
||||
* Every `memory_embeddings` row is stamped with the model that produced it (see
|
||||
* `runEmbedding` in `helpers.ts`). When the configured embedding model changes,
|
||||
* its vector dimension changes too, so the previously-stored vectors are no
|
||||
* longer comparable. On a mismatch we wipe every stored vector — the
|
||||
* `memory_embeddings` table, the `episodic_memory.binary_vector` column, and the
|
||||
* sqlite-vec `vec_episodes` index — then enqueue all live memories for
|
||||
* background re-embedding under the new model via `scheduleEmbedding`.
|
||||
*
|
||||
* Runs once per store open; a fresh store (no embeddings) or an already-current
|
||||
* store is a no-op. The destructive wipe is skipped whenever it could not be
|
||||
* rebuilt — embeddings disabled via the runtime option OR the
|
||||
* `MNEMOPI_NO_EMBEDDINGS` env, or an unresolved (empty) active model — so a
|
||||
* stale-but-valid corpus is never destroyed without a replacement. MUST run
|
||||
* inside the active runtime-options scope so `currentEmbeddingModel()` /
|
||||
* `embeddingsDisabled()` reflect the per-instance configuration.
|
||||
*/
|
||||
export function reconcileEmbeddingModel(beam: BeamMemoryState): void {
|
||||
if (embeddingsDisabled()) return;
|
||||
const active = currentEmbeddingModel().trim();
|
||||
if (active === "") return;
|
||||
|
||||
// Stop at the first row whose stamped model differs from the active one
|
||||
// (NULL/unstamped counts as a mismatch via `IS NOT`); an empty store or an
|
||||
// all-current store short-circuits here without gathering the DISTINCT set.
|
||||
const mismatch = beam.db.query("SELECT 1 FROM memory_embeddings WHERE model IS NOT ? LIMIT 1").get(active);
|
||||
if (!mismatch) return;
|
||||
|
||||
const staleModels = beam.db
|
||||
.query("SELECT DISTINCT model FROM memory_embeddings WHERE model IS NOT ?")
|
||||
.all(active) as { model: string | null }[];
|
||||
|
||||
const live = beam.db
|
||||
.query(`
|
||||
SELECT id AS memoryId, content FROM working_memory WHERE superseded_by IS NULL
|
||||
UNION ALL
|
||||
SELECT id AS memoryId, content FROM episodic_memory WHERE superseded_by IS NULL
|
||||
`)
|
||||
.all() as EmbedItem[];
|
||||
|
||||
transaction(beam.db, () => {
|
||||
beam.db.prepare("DELETE FROM memory_embeddings").run();
|
||||
beam.db.prepare("UPDATE episodic_memory SET binary_vector = NULL").run();
|
||||
if (vecAvailable(beam.db)) {
|
||||
try {
|
||||
beam.db.prepare("DELETE FROM vec_episodes").run();
|
||||
} catch {
|
||||
// sqlite-vec cleanup is best-effort; rebuild correctness takes precedence.
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
logger.info("mnemopi: embedding model changed, rebuilding", {
|
||||
from: staleModels.map(row => row.model ?? "(unstamped)"),
|
||||
to: active,
|
||||
count: live.length,
|
||||
});
|
||||
|
||||
// Re-embed in bounded batches so a corpus-wide rebuild never issues one giant
|
||||
// embedding request; each batch is its own tracked background task.
|
||||
for (let offset = 0; offset < live.length; offset += EMBED_REBUILD_BATCH) {
|
||||
scheduleEmbedding(beam, live.slice(offset, offset + EMBED_REBUILD_BATCH));
|
||||
}
|
||||
}
|
||||
|
||||
export function remember(beam: BeamMemoryState, content: string, options: StoreRememberOptions = {}): string {
|
||||
const source = options.source ?? "conversation";
|
||||
const importance = options.importance ?? 0.5;
|
||||
|
||||
@@ -99,7 +99,7 @@ function inTestRuntime(): boolean {
|
||||
return $env.NODE_ENV === "test" || $env.BUN_ENV === "test";
|
||||
}
|
||||
|
||||
function embeddingsDisabled(): boolean {
|
||||
export function embeddingsDisabled(): boolean {
|
||||
const active = activeEmbeddingOptions();
|
||||
if (active?.disabled !== undefined) {
|
||||
return active.disabled;
|
||||
|
||||
@@ -7,6 +7,7 @@ import type { MemoryInput, Metadata } from "../types";
|
||||
import { AnnotationStore } from "./annotations";
|
||||
import { BankManager } from "./banks";
|
||||
import { BeamMemory, initBeam } from "./beam/index";
|
||||
import { reconcileEmbeddingModel } from "./beam/store";
|
||||
import type { RecallEnhancedOptions, RecallOptions, RecallResult, SleepResult } from "./beam/types";
|
||||
import { EpisodicGraph } from "./episodic-graph";
|
||||
import {
|
||||
@@ -388,6 +389,10 @@ export class Mnemopi {
|
||||
}
|
||||
this.conn = this.beam.db;
|
||||
this.db = this.beam.db;
|
||||
// Wipe-and-rebuild stale embeddings when the configured model changed since
|
||||
// the vectors were written. Runs inside the runtime scope so
|
||||
// `currentEmbeddingModel()` reflects this instance's configured model.
|
||||
this.#withRuntimeOptions(() => reconcileEmbeddingModel(this.beam));
|
||||
}
|
||||
|
||||
close(): void {
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
/**
|
||||
* Wipe-and-rebuild reconcile: each `memory_embeddings` row is stamped with the
|
||||
* model that produced it. When the configured embedding model changes the vector
|
||||
* dimension changes too, so on store open `reconcileEmbeddingModel` (wired into
|
||||
* the `Mnemopi` constructor) wipes every stale vector and enqueues all live
|
||||
* memories for re-embedding under the new model. A matching model is a no-op.
|
||||
*/
|
||||
|
||||
import { Database } from "bun:sqlite";
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import "./setup";
|
||||
import { initBeam } from "@oh-my-pi/pi-mnemopi/core/beam";
|
||||
import { Mnemopi } from "@oh-my-pi/pi-mnemopi/core/memory";
|
||||
|
||||
const OLD_MODEL = "BAAI/bge-small-en-v1.5";
|
||||
const NEW_MODEL = "intfloat/multilingual-e5-large";
|
||||
|
||||
// Deterministic fastembed-shaped provider so the background rebuild actually
|
||||
// writes rows (and stamps them with the active model) under the test runtime.
|
||||
function fakeEmbed() {
|
||||
return async function* embed(texts: readonly string[]) {
|
||||
yield texts.map(() => [0.1, 0.2, 0.3, 0.4]);
|
||||
};
|
||||
}
|
||||
|
||||
function seedDb(model: string): { db: Database; ids: string[] } {
|
||||
const db = new Database(":memory:");
|
||||
initBeam(db);
|
||||
const ts = new Date().toISOString();
|
||||
db.prepare(
|
||||
"INSERT INTO working_memory (id, content, source, timestamp, session_id) VALUES (?, ?, 'test', ?, 'default')",
|
||||
).run("wm-1", "alpha working memory", ts);
|
||||
db.prepare(
|
||||
"INSERT INTO episodic_memory (id, content, source, timestamp, session_id, binary_vector) VALUES (?, ?, 'test', ?, 'default', ?)",
|
||||
).run("ep-1", "beta episodic memory", ts, new Uint8Array([1, 2, 3, 4]));
|
||||
for (const id of ["wm-1", "ep-1"]) {
|
||||
db.prepare("INSERT INTO memory_embeddings (memory_id, embedding_json, model) VALUES (?, ?, ?)").run(
|
||||
id,
|
||||
JSON.stringify([1, 0, 0, 0]),
|
||||
model,
|
||||
);
|
||||
}
|
||||
return { db, ids: ["wm-1", "ep-1"] };
|
||||
}
|
||||
|
||||
function countEmbeddings(memory: Mnemopi): number {
|
||||
return (memory.conn.query("SELECT COUNT(*) AS n FROM memory_embeddings").get() as { n: number }).n;
|
||||
}
|
||||
|
||||
describe("reconcileEmbeddingModel on store open", () => {
|
||||
it("wipes stale embeddings + binary vectors and re-embeds when the model changed", async () => {
|
||||
const { db, ids } = seedDb(OLD_MODEL);
|
||||
const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
|
||||
try {
|
||||
// Reconcile fired in the constructor: stale vector rows are gone and the
|
||||
// episodic binary vector was cleared. The async rebuild is enqueued but
|
||||
// has not yet run.
|
||||
expect(countEmbeddings(memory)).toBe(0);
|
||||
const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as {
|
||||
v: Uint8Array | null;
|
||||
};
|
||||
expect(ep.v).toBeNull();
|
||||
expect(memory.beam.pendingExtractions.size).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// The background rebuild repopulates every live memory, stamped with the
|
||||
// new model.
|
||||
await memory.flushExtractions();
|
||||
const rows = memory.conn.query("SELECT memory_id, model FROM memory_embeddings ORDER BY memory_id").all() as {
|
||||
memory_id: string;
|
||||
model: string;
|
||||
}[];
|
||||
expect(rows.map(row => row.memory_id).sort()).toEqual([...ids].sort());
|
||||
expect(rows.every(row => row.model === NEW_MODEL)).toBe(true);
|
||||
} finally {
|
||||
memory.close();
|
||||
db.close();
|
||||
}
|
||||
});
|
||||
|
||||
it("leaves embeddings untouched when the stored model already matches", () => {
|
||||
const { db } = seedDb(NEW_MODEL);
|
||||
const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
|
||||
try {
|
||||
// No mismatch -> no wipe, no rebuild enqueued.
|
||||
expect(countEmbeddings(memory)).toBe(2);
|
||||
expect(memory.beam.pendingExtractions.size).toBe(0);
|
||||
// The no-op path must preserve the episodic binary vector too (regression
|
||||
// guard against an unconditional clear that would silently drop vectors).
|
||||
const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as {
|
||||
v: Uint8Array | null;
|
||||
};
|
||||
expect(ep.v).not.toBeNull();
|
||||
expect(Array.from(ep.v as Uint8Array)).toEqual([1, 2, 3, 4]);
|
||||
} finally {
|
||||
memory.close();
|
||||
db.close();
|
||||
}
|
||||
});
|
||||
|
||||
it("does not wipe when embeddings are disabled via the MNEMOPI_NO_EMBEDDINGS env", () => {
|
||||
const { db } = seedDb(OLD_MODEL);
|
||||
const previous = process.env.MNEMOPI_NO_EMBEDDINGS;
|
||||
process.env.MNEMOPI_NO_EMBEDDINGS = "1";
|
||||
let memory: Mnemopi | undefined;
|
||||
try {
|
||||
// The model differs, but with embeddings disabled the rebuild would
|
||||
// produce nothing — so the stale-but-present vectors must survive.
|
||||
memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
|
||||
expect(countEmbeddings(memory)).toBe(2);
|
||||
expect(memory.beam.pendingExtractions.size).toBe(0);
|
||||
const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as {
|
||||
v: Uint8Array | null;
|
||||
};
|
||||
expect(ep.v).not.toBeNull();
|
||||
} finally {
|
||||
if (previous === undefined) {
|
||||
delete process.env.MNEMOPI_NO_EMBEDDINGS;
|
||||
} else {
|
||||
process.env.MNEMOPI_NO_EMBEDDINGS = previous;
|
||||
}
|
||||
memory?.close();
|
||||
db.close();
|
||||
}
|
||||
});
|
||||
|
||||
it("does not wipe when the active embedding model is empty", () => {
|
||||
const { db } = seedDb(OLD_MODEL);
|
||||
let memory: Mnemopi | undefined;
|
||||
try {
|
||||
// An explicit empty model resolves to no embedder; wiping would be
|
||||
// unrecoverable, so the reconcile must skip.
|
||||
memory = new Mnemopi({ db, embeddings: { model: "", provider: fakeEmbed() } });
|
||||
expect(countEmbeddings(memory)).toBe(2);
|
||||
expect(memory.beam.pendingExtractions.size).toBe(0);
|
||||
} finally {
|
||||
memory?.close();
|
||||
db.close();
|
||||
}
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user