diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 68c719a12..54e221a4d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -26,7 +26,7 @@ ### Fixed -- Stopped Mnemopi retention from overflowing the embedding model's context window: `embed()` now caps each input at `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS` (default 32000 chars) so a long multi-turn `MnemopiSessionState.retainMessages` transcript can't make llama.cpp's `/embeddings` server reject the request with `request (N tokens) exceeds the available context size` and silently drop vector recall for that memory ([#3126](https://github.com/can1357/oh-my-pi/issues/3126)). +- Stopped Mnemopi retention from overflowing the embedding model's context window: `embed()` now caps each input at `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS` (default 8192 chars) so a long multi-turn `MnemopiSessionState.retainMessages` transcript can't make llama.cpp's `/embeddings` server reject the request with `request (N tokens) exceeds the available context size` and silently drop vector recall for that memory ([#3126](https://github.com/can1357/oh-my-pi/issues/3126)). ## [16.1.7] - 2026-06-20 diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index dfb11763d..ba3a5658c 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Capped per-input length in `embed()` at `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS` (default 32000 chars, override via the env var or `embeddings.maxInputChars` runtime option; `0` disables) so a long retention transcript can no longer overflow the embedding model's context window. llama.cpp's `/embeddings` server used to reject the request with `request (N tokens) exceeds the available context size`, silently dropping vector recall for that memory ([#3126](https://github.com/can1357/oh-my-pi/issues/3126)). +- Capped per-input length in `embed()` at `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS` (default 8192 chars, override via the env var or `embeddings.maxInputChars` runtime option; `0` disables) so a long retention transcript can no longer overflow the embedding model's context window. llama.cpp's `/embeddings` server used to reject the request with `request (N tokens) exceeds the available context size`, silently dropping vector recall for that memory ([#3126](https://github.com/can1357/oh-my-pi/issues/3126)). ## [16.1.3] - 2026-06-19 diff --git a/packages/mnemopi/src/config.ts b/packages/mnemopi/src/config.ts index 3bee5a23c..56ee807ca 100644 --- a/packages/mnemopi/src/config.ts +++ b/packages/mnemopi/src/config.ts @@ -110,13 +110,13 @@ export function embeddingsDisabled(env: Env = process.env): boolean { * source gives both backends deterministic behavior and prevents the silent * recall degradation we saw in issue #3126. * - * Default `32000` ≈ 8k English tokens (4 chars/token) or ~16k–32k CJK tokens — - * fits 8k-context models (bge-m3, text-embedding-3) for English and most CJK - * content. Lower it for 512-token models like `BAAI/bge-small-en-v1.5`, raise - * it for larger contexts. `0` disables the cap. + * Default `8192` chars is intentionally conservative for 8192-token embedding + * contexts (bge-m3, OpenAI text-embedding-3) and CJK-heavy transcripts. Raise + * it for larger local contexts (for example Qwen3-Embedding with 32k ctx). + * `0` disables the cap. */ export function embeddingMaxInputChars(env: Env = process.env): number { - return Math.max(0, envInt("MNEMOPI_EMBEDDING_MAX_INPUT_CHARS", 32000, env)); + return Math.max(0, envInt("MNEMOPI_EMBEDDING_MAX_INPUT_CHARS", 8192, env)); } export function isApiEmbeddingModel(model = embeddingModel(), env: Env = process.env): boolean { diff --git a/packages/mnemopi/src/core/embeddings.ts b/packages/mnemopi/src/core/embeddings.ts index 37f79b10c..2d84ca7a1 100644 --- a/packages/mnemopi/src/core/embeddings.ts +++ b/packages/mnemopi/src/core/embeddings.ts @@ -124,7 +124,7 @@ export function embeddingsDisabled(): boolean { * Resolved per-input character cap for {@link embed}. * * Reads (in order): the active runtime scope's `embeddings.maxInputChars`, then - * `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS`, then the bundled `32000` default. `0` + * `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS`, then the bundled `8192` default. `0` * disables the cap entirely. */ function effectiveMaxInputChars(): number { @@ -132,7 +132,7 @@ function effectiveMaxInputChars(): number { if (override !== undefined) return Math.max(0, Math.trunc(override)); const envValue = Number.parseInt($env.MNEMOPI_EMBEDDING_MAX_INPUT_CHARS ?? "", 10); if (Number.isFinite(envValue) && envValue >= 0) return envValue; - return 32000; + return 8192; } /** diff --git a/packages/mnemopi/test/embedding-input-cap.test.ts b/packages/mnemopi/test/embedding-input-cap.test.ts index 050e729be..46217ac3c 100644 --- a/packages/mnemopi/test/embedding-input-cap.test.ts +++ b/packages/mnemopi/test/embedding-input-cap.test.ts @@ -13,7 +13,7 @@ import { withMnemopiRuntimeOptions } from "@oh-my-pi/pi-mnemopi/core/runtime-opt * overflow whatever ctx the embedding server was started with — llama.cpp * rejects oversized requests with `request (N tokens) exceeds the available * context size`, OpenAI silently right-truncates. `embed()` now caps each - * input to `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS` (default 32000) before the + * input to `MNEMOPI_EMBEDDING_MAX_INPUT_CHARS` (default 8192) before the * provider sees it. */ function captureProvider(): { @@ -57,7 +57,7 @@ describe("embed() input cap (#3126)", () => { expect(provider.calls).toHaveLength(1); const [seenShort, seenHuge] = provider.calls[0] ?? []; expect(seenShort).toBe("short"); - expect(seenHuge?.length).toBe(32_000); + expect(seenHuge?.length).toBe(8192); }); it("honors MNEMOPI_EMBEDDING_MAX_INPUT_CHARS env override", async () => {