From c4c033134519c57c241f4777edbae755e7911036 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 2 Jul 2026 03:57:35 +0200 Subject: [PATCH] fix(coding-agent/session): prevented data loss in session serialization - Persist signed message blocks (`text`, `thinking`, `toolCall`) and encrypted reasoning payloads verbatim during session serialization instead of clearing or truncating them. - Preserve signature keys instead of replacing them with empty strings when they exceed persistence size limits. - Exempt official first-party OpenAI and Anthropic API endpoints from the leaked-thinking stream healing wrapper to prevent misfires on legitimate visible text fences. --- packages/agent/CHANGELOG.md | 3 +- packages/ai/CHANGELOG.md | 17 ++--- packages/ai/src/stream.ts | 3 +- .../ai/src/utils/leaked-thinking-stream.ts | 12 ++- .../ai/test/leaked-thinking-stream.test.ts | 12 ++- .../ai/test/stream-markup-healing.test.ts | 29 ++++++++ packages/catalog/CHANGELOG.md | 7 +- packages/catalog/src/compat/openai.ts | 31 ++++++-- packages/catalog/test/build.test.ts | 24 ++++++ packages/coding-agent/CHANGELOG.md | 74 +++---------------- .../src/session/session-persistence.ts | 45 ++++++----- .../signature-persistence.test.ts | 39 +--------- packages/collab-web/CHANGELOG.md | 4 +- packages/hashline/CHANGELOG.md | 2 +- packages/natives/CHANGELOG.md | 2 +- packages/tui/CHANGELOG.md | 2 +- packages/wire/CHANGELOG.md | 4 +- 17 files changed, 158 insertions(+), 152 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 2ef0416f3..16504b27c 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -8,8 +8,7 @@ ### Fixed -- Fixed an issue where legacy steering messages were prematurely consumed and dropped during in-flight tool execution polls - +- Fixed an issue where legacy steering messages were prematurely consumed and dropped during in-flight tool execution polls. - Fixed an issue where skipped tool results in queued messages were incorrectly treated as completed, preventing necessary retries. - Improved branch summaries to preserve informative tool results from abandoned branches while filtering out redundant output. - Fixed interruptible tool waits to properly abort on host-provided IRC interrupts in addition to user steering. diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 1c0a723d8..f0ee21eeb 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,21 +4,20 @@ ### Added -- Added opt-in support for Anthropic's server-side fallback beta (`server-side-fallback-2026-06-01`) on the `anthropic-messages` provider, including support for `AnthropicOptions.fallbacks`, mid-stream fallback content blocks, fallback billing/usage iterations, and automatic filtering of fallback blocks during cross-provider message transformations. +- Added opt-in support for Anthropic's server-side fallback beta (server-side-fallback-2026-06-01) on the anthropic-messages provider, including support for AnthropicOptions.fallbacks and automatic filtering of fallback blocks during cross-provider message transformations. ### Changed -- Skip leaked-thinking stream healing for official first-party Anthropic, OpenAI, and Codex endpoints - -- Updated CoreWeave Serverless Inference login instructions to clarify persisting `COREWEAVE_PROJECT` in shell startup files. +- Improved stream healing for official first-party endpoints (Anthropic, OpenAI, and OpenAI Codex) by skipping leaked-thinking healing, preventing misfires on legitimate code blocks while maintaining healing for third-party gateways and custom base URLs. +- Updated CoreWeave Serverless Inference login instructions to clarify persisting COREWEAVE_PROJECT in shell startup files. ### Fixed -- Fixed an issue where broker usage fetch failures were not cached, causing redundant network requests during sequential ranking passes when the broker is offline. -- Fixed Xiaomi MiMo API key validation to use the supported `mimo-v2.5` model. -- Fixed certificate verification errors for custom gateways behind private CA bundles by ensuring `NODE_EXTRA_CA_CERTS` is respected across all provider fetches (including OpenAI-compatible, Codex, Ollama, Azure, and Google). -- Fixed Claude Fable demoted-thinking replay to use markdown-italic assistant prose instead of `` tags, preventing context issues after model switches. -- Fixed OpenAI Responses replay errors (400 Bad Request) caused by missing reasoning items in locally rebuilt assistant item IDs during history replay. +- Fixed a performance issue where broker usage fetch failures were not cached, causing redundant network requests when the broker is offline. +- Fixed Xiaomi MiMo API key validation to use the supported mimo-v2.5 model. +- Fixed certificate verification errors for custom gateways behind private CA bundles by ensuring NODE_EXTRA_CA_CERTS is respected across all provider fetches. +- Fixed Claude Fable demoted-thinking replay to use markdown-italic assistant prose instead of tags, preventing context issues after model switches. +- Fixed OpenAI Responses replay errors (400 Bad Request) caused by missing reasoning items during history replay. ## [16.2.13] - 2026-07-01 diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 85a39c1b4..3d029963f 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -99,7 +99,8 @@ function isGoogleVertexAuthenticatedModel(model: Model): boolean { * gateway may well leak. URL checks are strict (exact origin / path boundary * or parsed hostname) — a substring match would accept lookalikes like * `https://api.openai.com.evil/`. Anthropic Foundry (`CLAUDE_CODE_USE_FOUNDRY`) - * dispatches an empty `baseUrl` to an enterprise gateway, so it is never exempt. + * redirects an empty `baseUrl` to `FOUNDRY_BASE_URL`, so the check runs against + * that effective endpoint — exempt only when it resolves to the official host. */ function isLeakedThinkingHealExempt(model: Model): boolean { switch (model.provider) { diff --git a/packages/ai/src/utils/leaked-thinking-stream.ts b/packages/ai/src/utils/leaked-thinking-stream.ts index 8647338e4..074a4c507 100644 --- a/packages/ai/src/utils/leaked-thinking-stream.ts +++ b/packages/ai/src/utils/leaked-thinking-stream.ts @@ -3,11 +3,15 @@ * * Some providers emit their canonical reasoning idioms (` ```thinking `, * ``, Gemma/Harmony channels, …) into the *visible* text stream instead - * of a structured thinking part. {@link wrapLeakedThinkingStream} re-projects any + * of a structured thinking part. {@link wrapLeakedThinkingStream} re-projects a * provider stream into a fresh {@link AssistantMessageEventStream}, splitting the - * leaked fences out into proper `thinking` blocks *live* as deltas arrive — so - * every provider gets the same healing, not just the three with provider-local - * {@link StreamMarkupHealing} loops. + * leaked fences out into proper `thinking` blocks *live* as deltas arrive. + * + * Applied to every provider stream *except* official first-party endpoints + * (the official Anthropic API and the official OpenAI / OpenAI-Codex endpoints), + * which return structured thinking and never leak — `healLeakedThinking` in + * `../stream.ts` gates the wrap so the healer cannot misfire on legitimate + * fenced content those models emit as visible text. * * The healing is idempotent: a second pass over already-clean text finds no * fences, so wrapping a provider that already heals (or wrapping twice) is a diff --git a/packages/ai/test/leaked-thinking-stream.test.ts b/packages/ai/test/leaked-thinking-stream.test.ts index 9911745b8..fadd62c0f 100644 --- a/packages/ai/test/leaked-thinking-stream.test.ts +++ b/packages/ai/test/leaked-thinking-stream.test.ts @@ -446,10 +446,14 @@ describe("leaked thinking healing through stream()", () => { it("splits a leaked fence for a non-official anthropic-messages endpoint", async () => { // A third-party gateway reusing the anthropic-messages wire format may leak, // so the central wrapper still heals when the endpoint is not official. - const result = await stream(anthropicModel({ provider: "z-ai", baseUrl: "https://api.z.ai/api/anthropic" }), context, { - apiKey: "test", - fetch: anthropicLeakFetch(leaked), - }).result(); + const result = await stream( + anthropicModel({ provider: "zai", baseUrl: "https://api.z.ai/api/anthropic" }), + context, + { + apiKey: "test", + fetch: anthropicLeakFetch(leaked), + }, + ).result(); expect(result.content.map(b => b.type)).toEqual(["thinking", "text"]); const thinking = thinks(result) diff --git a/packages/ai/test/stream-markup-healing.test.ts b/packages/ai/test/stream-markup-healing.test.ts index 23f525eeb..3401e058b 100644 --- a/packages/ai/test/stream-markup-healing.test.ts +++ b/packages/ai/test/stream-markup-healing.test.ts @@ -246,6 +246,35 @@ describe("openai-completions leaked thinking healing", () => { }); }); +describe("official OpenAI leaked thinking healing exemption", () => { + // The official OpenAI endpoint returns structured reasoning and never leaks + // fences, so neither the provider-local healer nor the central + // wrapLeakedThinkingStream wrap runs — a ` ```thinking ` block the model chose + // to write must stay verbatim visible text. Routed through stream() (not the + // provider directly) so both gates are exercised. + const officialOpenAI = getBundledModel("openai", "gpt-5.5"); + const completionsModel = buildModel({ + ...officialOpenAI, + api: "openai-completions", + }); + + it("leaves a leaked fence intact for the official OpenAI endpoint", async () => { + const leaked = "```thinking\nWeigh the options.\n```\nFinal answer."; + const result = await stream(completionsModel, baseContext(), { + apiKey: "test", + fetch: mockFetch([ + chunk(completionsModel.id, { content: leaked }), + chunk(completionsModel.id, {}, "stop"), + "[DONE]", + ]), + }).result(); + + expect(result.content.map(b => b.type)).toEqual(["text"]); + expect(result.content.filter((b): b is ThinkingContent => b.type === "thinking")).toHaveLength(0); + expect(result.content.map(b => (b.type === "text" ? b.text : "")).join("")).toBe(leaked); + }); +}); + describe("google-gemini-cli leaked thinking healing", () => { it("lifts a leaked Gemini thinking fence before a native tool call", async () => { const model = geminiCliModel(); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 9f0ca4296..01c9504b8 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -8,10 +8,11 @@ ### Fixed -- Fixed the Xiaomi provider's default model to use the supported `mimo-v2.5` model. -- Fixed model discovery probes (including Ollama and metadata fetches) failing behind private-CA gateways by ensuring they honor `NODE_EXTRA_CA_CERTS`. +- Fixed stream markup healing pattern misfires by disabling the healer on the official OpenAI endpoint. +- Updated the Xiaomi provider's default model to the supported `mimo-v2.5` model. +- Fixed model discovery probes (including Ollama and metadata fetches) failing behind private-CA gateways by ensuring they honor the `NODE_EXTRA_CA_CERTS` environment variable. - Fixed CoreWeave Serverless Inference project-header detection to ensure blank OpenAI-Project overrides do not block the `COREWEAVE_PROJECT` fallback. -- Fixed LiteLLM MiniMax M3 discovery to remove reseller-only display suffixes, and invalidated the model cache to ensure stale suffixes are cleared immediately. +- Fixed LiteLLM MiniMax M3 discovery to remove reseller-only display suffixes and invalidated the model cache to clear stale suffixes immediately. ## [16.2.13] - 2026-07-01 diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index b337791d7..6de9b16d4 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -117,20 +117,39 @@ function resolveReasoningDisableMode( /** * Pick the leaked-markup healer for an OpenAI-compatible visible-text stream. * Kimi chat-template tokens and DeepSeek DSML envelopes need their dedicated - * tool-call grammars; every other model defaults to `"thinking"`. All patterns - * run the generic thinking healer, so leaked reasoning idioms (e.g. a Gemini - * ` ```thinking ` fence on OpenRouter) are always recovered from `delta.content`. + * tool-call grammars. Every other OpenAI-compatible model defaults to + * `"thinking"` so leaked reasoning idioms (e.g. a Gemini ` ```thinking ` fence + * on OpenRouter) are recovered from `delta.content` — **except** the official + * OpenAI endpoint (`provider: "openai"` + `api.openai.com`), which returns + * structured reasoning and never leaks, so it heals nothing (returns + * `undefined`) to avoid misfiring on legitimate fenced content. */ -function detectStreamMarkupHealingPattern(provider: string, modelId: string): OpenAIStreamMarkupHealingPattern { +function detectStreamMarkupHealingPattern( + provider: string, + modelId: string, + baseUrl: string, +): OpenAIStreamMarkupHealingPattern | undefined { if (provider === "kimi-code" || provider === "moonshot" || /kimi[-/_.]?k2/i.test(modelId)) { return "kimi"; } if (isDeepseekModelIdOrName(modelId) && DSML_HEALING_PROVIDERS.has(provider)) { return "dsml"; } + if (isOfficialOpenAIEndpoint(provider, baseUrl)) return undefined; return "thinking"; } +/** Strict official-OpenAI check: provider id `openai` and an `api.openai.com` host (missing baseUrl defaults there). */ +function isOfficialOpenAIEndpoint(provider: string, baseUrl: string): boolean { + if (provider !== "openai") return false; + if (!baseUrl) return true; + try { + return new URL(baseUrl).hostname === "api.openai.com"; + } catch { + return false; + } +} + /** * OpenCode's gateways (https://opencode.ai/zen|go) gate `reasoning_content` * on the request's thinking state for every model they front (Kimi K2.x, @@ -517,7 +536,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv streamIdleTimeoutMs, stripDeepseekSpecialTokens: isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"), - streamMarkupHealingPattern: detectStreamMarkupHealingPattern(provider, spec.id), + streamMarkupHealingPattern: detectStreamMarkupHealingPattern(provider, spec.id, baseUrl), reasoningDeltasMayBeCumulative: MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.id), emptyLengthFinishIsContextError: provider === "ollama", @@ -646,7 +665,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol supportsObfuscationOptOut: isOpenAIUrl || spec.provider === "openai", stripDeepseekSpecialTokens: Boolean(id) && isDeepseekModelIdOrName(id) && (spec.provider === "nvidia" || spec.provider === "deepseek"), - streamMarkupHealingPattern: id ? detectStreamMarkupHealingPattern(spec.provider, id) : undefined, + streamMarkupHealingPattern: id ? detectStreamMarkupHealingPattern(spec.provider, id, baseUrl) : undefined, reasoningDeltasMayBeCumulative: MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.provider) || (id ? MINIMAX_PROVIDER_OR_ID_PATTERN.test(id) : false), emptyLengthFinishIsContextError: spec.provider === "ollama", diff --git a/packages/catalog/test/build.test.ts b/packages/catalog/test/build.test.ts index b3f7c3300..706321869 100644 --- a/packages/catalog/test/build.test.ts +++ b/packages/catalog/test/build.test.ts @@ -297,6 +297,30 @@ describe("openai-completions wire-quirk compat detection", () => { expect(buildOpenAICompat(completionsSpec()).dropThinkingWhenReasoningEffort).toBe(false); }); + it("disables the leaked-markup healer for the official OpenAI endpoint only", () => { + // Official OpenAI returns structured reasoning and never leaks fences, so + // the provider-local healer stays off; every other OpenAI-compatible host + // keeps the default "thinking" healer, and Kimi/DSML keep their grammars. + expect( + buildOpenAICompat(completionsSpec({ provider: "openai", baseUrl: "https://api.openai.com/v1" })) + .streamMarkupHealingPattern, + ).toBeUndefined(); + expect( + buildOpenAICompat(completionsSpec({ provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1" })) + .streamMarkupHealingPattern, + ).toBe("thinking"); + // A lookalike host under the openai provider id is NOT the official endpoint. + expect( + buildOpenAICompat(completionsSpec({ provider: "openai", baseUrl: "https://api.openai.com.evil/v1" })) + .streamMarkupHealingPattern, + ).toBe("thinking"); + expect( + buildOpenAICompat( + completionsSpec({ provider: "moonshot", id: "kimi-k2", baseUrl: "https://api.moonshot.ai/v1" }), + ).streamMarkupHealingPattern, + ).toBe("kimi"); + }); + it("derives Responses obfuscation opt-out and wire mode per surface", () => { expect( buildOpenAIResponsesCompat({ diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a348f47de..0954cd509 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added `providers.anthropic.serverSideFallback` configuration option to opt into Anthropic's server-side-fallback beta chain, allowing Claude Fable 5 / Mythos 5 requests to automatically retry on Opus 4.8 when blocked by classifiers. +- Added `providers.anthropic.serverSideFallback` configuration option to opt into Anthropic's server-side-fallback beta chain, allowing Claude requests to automatically retry on alternative models when blocked by classifiers. - Added `task.softRequestBudgetNotice` configuration option to enable subagent soft-budget wrap-up steering notices while keeping the graceful abort guard active. ### Changed @@ -16,66 +16,7 @@ ### Fixed -- Fixed subagent HUD layout and tree rendering in the TUI to align correctly with other HUD panels -- Fixed session persistence corrupting Anthropic signed thinking blocks: the generic 500K-character truncation cap could shorten a large `thinking` string while leaving its cryptographic signature intact, so replaying or resuming the session sent a signature that no longer matched the text and the provider rejected it with an HTTP 400. Signed `thinking` blocks and encrypted `redactedThinking` blobs are now persisted verbatim (all-or-nothing); unsigned thinking and plain text remain truncatable for size control. -- Fixed post-rewind context to tell the agent the checkpoint completed and to make repeat `rewind` calls recover with guidance instead of a bare no-checkpoint error; the branch-scan rehydration now also restores an active checkpoint when a session resumes with the latest checkpoint not yet rewound, so `rewind` can complete instead of failing with "No active checkpoint". ([#4187](https://github.com/can1357/oh-my-pi/issues/4187)) -- Fixed task.maxConcurrency being breachable when a queued spawn was cancelled: the spawn path could release a semaphore permit it never acquired, letting a later task start while the cap was saturated. -- Fixed session exit diagnostics recording signal and crash exits (SIGTERM, SIGHUP, uncaught exceptions) as a normal "dispose": the postmortem teardown now threads the real reason into session disposal. -- Fixed the subagent yield-label guard ignoring JTD discriminator (oneOf) output schemas, which let stale incremental labels pass into successful results when final validation was skipped after retries. -- Fixed grep/ast_grep search scopes rejecting `www.` and collapsed-scheme (`https:/host`) URL spellings that the read tool accepts; unresolvable URL-shaped scopes now fail with an explicit external-URL error instead of "Path not found". -- Fixed model discovery ignoring `NODE_EXTRA_CA_CERTS`: the model registry's default fetch now applies the extra-CA wrapper, so `/models` probes work behind private-CA gateways like provider chat requests. -- Fixed ctrl+p role-model cycling getting stuck on one transition and skipping every other role: a session-branch traversal regression returned entries leaf-to-root, so the cycle (and session model restore) read the oldest recorded model change instead of the newest. -- Fixed ctrl+p cycling from a stale slot after the model was switched through another surface (alt+m, /model, retry fallback): the recorded role is now trusted only while its resolved model is still the active model, falling back to matching by model. -- Fixed the apply_patch tool to prevent silently overwriting pre-existing files during creation or renaming, rejecting upfront with an error instead. -- Fixed multi-file apply_patch to stop at the first failing file, surface applied vs. skipped paths, and correctly report the error to the agent loop. -- Fixed process termination (SIGTERM, SIGHUP, uncaught exceptions) skipping editor draft saves, session shutdown events, and background job cleanup. -- Fixed /quit and /exit commands blocking session closure by introducing a shutdown budget and backgrounding remaining tasks. -- Fixed git and GitHub CLI subprocesses hanging on interactive prompts by forcing non-interactive environments, adding timeouts, and capping output. -- Fixed RPC mode abort_bash being blocked by running bash commands by dispatching bash in the background. -- Fixed task.maxConcurrency and task.maxRecursionDepth limits being bypassed by sub-spawn paths, ensuring limits are dynamically resized and respected. -- Fixed the edit tool inflating session files by pruning extremely large file snapshots from tool-result details. -- Fixed edit-tool Markdown list guidance so hashline parser errors and the model-facing prompt teach `+- item` escaping instead of steering agents toward full-file `write` fallbacks. ([#4179](https://github.com/can1357/oh-my-pi/issues/4179)) -- Fixed workstation OS detection rendering "Kernel: unknown" on macOS 15+. -- Fixed /copy code and /copy cmd commands being treated as normal prompts instead of copying the requested blocks. -- Fixed interactive bash status line not updating after directory changes (cd). -- Fixed session title refreshes ignoring user TITLE_SYSTEM.md overrides during replans, and prevented auto-generated titles from incorrectly preserving all-caps text from user messages. -- Fixed the live todo HUD going stale during long tool-use loops by adding mid-run reminders for incomplete items. -- Fixed /shake and mid-stream chat rebuilds erasing active LLM output. -- Fixed RpcClient failing to restart after being stopped or failing on initial startup. -- Fixed /collab web guests being unable to answer ask tool questions by routing host UI requests through writable collab peers. -- Fixed search and AST tools accepting external read URLs by materializing fetched URL text through the read cache before path resolution. -- Fixed Tavily web search to retry without recency filters if no content is returned. -- Fixed extension validation failures for omp install pi-lean-ctx by exposing legacy tool factories. -- Fixed visibility of the focused option in the multi-select ask picker on certain color themes. -- Fixed TUI row overlapping and duplication in the eval tool's live subagent progress tree under heavy concurrency. -- Fixed session resumes after silent exits by recording pre-tool start markers and shutdown diagnostics. -- Fixed status-line redraw crashes when tool-call arguments contain BigInt values. -- Fixed terminal scrollback rows retaining old colors after theme switches. -- Fixed Esc key behavior in the TUI to clear unrecoverable input instead of preserving drafts. -- Fixed marketplace-installed plugins appearing redundantly in both the npm list and the extension status provider. -- Fixed parent and peer IRC message delivery delays. -- Fixed provider/model:auto entries in modelRoles collapsing to inherit and losing their auto state on reload. -- Fixed plan execution prompts to avoid embedding the full plan, referencing the local plan file instead. -- Fixed subagent live progress leaking raw tool output into the parent TUI. -- Fixed eval subagents with custom output schemas receiving stale incremental yield labels. -- Fixed hidden-thinking live status rows rendering as glyph-only lines by adding a persistent label. -- Fixed /compact summary divider placement to keep it in the live scrollable region. -- Fixed the time_spent status-line segment ticking continuously during idle sessions. -- Fixed browser tool schema validation to require the code argument for run calls. -- Improved robustness of MCP authentication error detection and header-based server discovery. -- Fixed reliable detection of 401/403 authorization failures during Smithery commands and HTTP RPCs. -- Improved streaming preview responsiveness for write, edit, and eval tools by decoding streamed string arguments incrementally. -- Added retry-path diagnostics for assistant-tail removal and scheduled continuations after transient provider errors. -- Fixed CJK history rendering issues across repeated compactions. -- Fixed user-invoked skills failing to identify themselves or resolve relative paths across various execution paths. -- Fixed type errors introduced by the merge sweep: restored the ask row-budget priority field, narrowed dereferenced schema property access, and updated stale test API usage. -- Fixed git clone and fetch being killed by the 5-minute local-command timeout; network transfers now use a separate 30-minute deadline, overridable per call. -- Fixed the TUI collab guest (omp join) silently dropping host ask/selector UI requests; they now present through the standard dialog flow and round-trip responses, with cancellation and resync replay handled. -- Fixed transcript rebuilds (theme change, /shake, focus replay) showing stale streamed write/edit/eval content by sharing the partial-JSON decode between the live streaming path and every rebuild path. -- Fixed an explicitly configured compaction.reserveTokens equal to the built-in default being silently replaced by the proportional small-window fallback; the setting now defaults to unset and explicit values are always honored. -- Fixed user-configured LiteLLM discovery providers keeping stale reseller display-name suffixes for up to 24 hours after upgrade by invalidating the warm model cache. -### Fixed - +- Fixed session persistence corrupting Anthropic, OpenAI, and Google signed thinking blocks and reasoning payloads, ensuring cryptographic signatures and encrypted content are preserved verbatim to prevent API replay rejections (HTTP 400). - Fixed several issues with the `apply_patch` and edit tools, including preventing dirty buffers on early aborts, rejecting overwrites of pre-existing files, stopping at the first failing file in multi-file operations, and pruning extremely large file snapshots to prevent session inflation. - Fixed process termination (SIGTERM, SIGHUP, uncaught exceptions) to ensure editor drafts are saved, sessions shut down cleanly, and background jobs are cleaned up. - Fixed `/quit` and `/exit` commands blocking session closure by introducing a shutdown budget and backgrounding remaining tasks. @@ -89,7 +30,16 @@ - Fixed `/copy code` and `/copy cmd` commands being treated as normal prompts instead of copying the requested blocks. - Fixed legacy tool compatibility issues for `createReadTool`, `createGrepTool`, and extension validation failures for `omp install pi-lean-ctx`. - Improved robustness of MCP authentication error detection, Smithery command authorization failures, and streaming preview responsiveness for write, edit, and eval tools. -- Fixed llama.cpp router/preset mode reporting 128k context in the status bar for every preset picked from `/model` regardless of the preset's configured `--ctx-size`. The router-level `/v1/models` only carries `meta.n_ctx` after a preset's child instance is loaded, and router `/props` reports a dummy `n_ctx: 0`, so cold-picked presets fell through to the 128k discovery default. Discovery now also reads `--ctx-size` (or `-c`) from each entry's `status.args` rendered CLI vector and, as a fallback, `ctx-size = N` from `status.preset` INI; the same fallback applies on the runtime refresh that runs when `/model` switches to a preset ([#4190](https://github.com/can1357/oh-my-pi/issues/4190)). +- Fixed llama.cpp router/preset mode reporting 128k context in the status bar for every preset picked from `/model` regardless of the preset's configured `--ctx-size`. +- Fixed post-rewind context to tell the agent the checkpoint completed and to make repeat `rewind` calls recover with guidance instead of a bare no-checkpoint error. +- Fixed grep/ast_grep search scopes rejecting `www.` and collapsed-scheme (`https:/host`) URL spellings. +- Fixed ctrl+p role-model cycling getting stuck on one transition and skipping every other role. +- Fixed interactive bash status line not updating after directory changes (cd). +- Fixed session title refreshes ignoring user `TITLE_SYSTEM.md` overrides during replans, and prevented auto-generated titles from incorrectly preserving all-caps text from user messages. +- Fixed the live todo HUD going stale during long tool-use loops by adding mid-run reminders for incomplete items. +- Fixed `/shake` and mid-stream chat rebuilds erasing active LLM output. +- Fixed Tavily web search to retry without recency filters if no content is returned. +- Fixed user-configured LiteLLM discovery providers keeping stale reseller display-name suffixes for up to 24 hours after upgrade by invalidating the warm model cache. ## [16.2.13] - 2026-07-01 diff --git a/packages/coding-agent/src/session/session-persistence.ts b/packages/coding-agent/src/session/session-persistence.ts index 71b7ecfaf..68a2e3455 100644 --- a/packages/coding-agent/src/session/session-persistence.ts +++ b/packages/coding-agent/src/session/session-persistence.ts @@ -59,9 +59,15 @@ function shouldExternalizeImagePayload( return (key === TEXT_CONTENT_KEY && isImageBlock(value)) || key === "images"; } +/** True for a non-empty string — marks signature/encrypted fields whose block must persist verbatim. */ +function isNonEmptyString(value: unknown): value is string { + return typeof value === "string" && value.length > 0; +} + /** * Recursively truncate large strings in an object for session persistence. - * - Truncates any oversized string fields (key-agnostic) + * - Truncates oversized string fields (key-agnostic), except signed/encrypted + * blocks and signature keys, which persist verbatim * - Externalizes oversized image payloads to blob refs * - Updates lineCount when content is truncated * - Returns original object if no changes needed (structural sharing) @@ -76,20 +82,23 @@ function truncateForPersistence(obj: unknown, blobStore: BlobStore, key?: string if (shouldExternalizeImagePayload(obj, key)) { return { ...obj, data: externalizeImageDataSync(blobStore, obj.data, obj.mimeType) }; } - // Signed/encrypted reasoning is bound to its exact bytes: a truncated `thinking` - // no longer matches its signature and a truncated `redacted_thinking` blob is - // undecryptable, so the provider 400s the replay. Persist these verbatim — never - // truncate, externalize, or descend. Unsigned thinking (e.g. an interrupted - // stream) has no such binding and stays truncatable for size control. + // Signed content is bound to its exact bytes: a truncated `thinking`/`text`/ + // `arguments` no longer matches its signature and a truncated + // `redacted_thinking` blob is undecryptable, so the provider 400s the replay. + // Persist signed blocks verbatim — never truncate, externalize, or descend. + // Unsigned blocks (e.g. an interrupted stream) have no such binding and stay + // truncatable for size control. if (typeof obj === "object" && "type" in obj) { - const signedThinking = - obj.type === "thinking" && - "thinkingSignature" in obj && - typeof obj.thinkingSignature === "string" && - obj.thinkingSignature.length > 0; - const redacted = - obj.type === "redactedThinking" && "data" in obj && typeof obj.data === "string" && obj.data.length > 0; - if (signedThinking || redacted) return obj; + const signed = + (obj.type === "thinking" && "thinkingSignature" in obj && isNonEmptyString(obj.thinkingSignature)) || + (obj.type === "text" && "textSignature" in obj && isNonEmptyString(obj.textSignature)) || + (obj.type === "toolCall" && "thoughtSignature" in obj && isNonEmptyString(obj.thoughtSignature)); + const redacted = obj.type === "redactedThinking" && "data" in obj && isNonEmptyString(obj.data); + // OpenAI Responses reasoning items (providerPayload.items) carry + // `encrypted_content`, server-validated on replay — atomic like signed blocks. + const encryptedReasoning = + obj.type === "reasoning" && "encrypted_content" in obj && isNonEmptyString(obj.encrypted_content); + if (signed || redacted || encryptedReasoning) return obj; } if (typeof obj === "string") { @@ -97,10 +106,12 @@ function truncateForPersistence(obj: unknown, blobStore: BlobStore, key?: string return externalizeImageDataUrlSync(blobStore, obj); } if (obj.length > MAX_PERSIST_CHARS) { - // Cryptographic signatures must be preserved exactly or cleared entirely — never truncated. - // Truncation would produce an invalid signature that the API rejects. + // Defensive: signature keys normally sit on blocks the guard above returns + // verbatim, but if one is reached here (unknown carrier shape), preserve it — + // truncation produces an invalid signature the API rejects, and clearing + // drops reasoning context the provider needs on replay. if (key === "thinkingSignature" || key === "thoughtSignature" || key === "textSignature") { - return ""; + return obj; } const limit = Math.max(0, MAX_PERSIST_CHARS - TRUNCATION_NOTICE.length); return `${truncateString(obj, limit)}${TRUNCATION_NOTICE}`; diff --git a/packages/coding-agent/test/session-manager/signature-persistence.test.ts b/packages/coding-agent/test/session-manager/signature-persistence.test.ts index f4de85100..09c62798c 100644 --- a/packages/coding-agent/test/session-manager/signature-persistence.test.ts +++ b/packages/coding-agent/test/session-manager/signature-persistence.test.ts @@ -27,42 +27,6 @@ function getAssistantMessage(session: SessionManager): AssistantMessage { } describe("SessionManager signature persistence", () => { - it("clears oversized signatures instead of truncating them", async () => { - using tempDir = TempDir.createSync("@pi-session-signature-persistence-"); - const session = SessionManager.create(tempDir.path(), tempDir.path()); - - session.appendMessage({ role: "user", content: "continue", timestamp: 1 }); - session.appendMessage({ - role: "assistant", - content: [ - { type: "thinking", thinking: "reasoning", thinkingSignature: "s".repeat(600_000) }, - { type: "text", text: "done", textSignature: "m".repeat(600_000) }, - { type: "toolCall", id: "tool_1", name: "read", arguments: {}, thoughtSignature: "t".repeat(600_000) }, - ], - api: "openai-responses", - provider: "openai", - model: "gpt-5-mini", - usage: { - input: 1, - output: 1, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 2, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "stop", - timestamp: 2, - } satisfies AssistantMessage); - await session.flush(); - - const reloaded = await SessionManager.open(session.getSessionFile()!); - const assistant = getAssistantMessage(reloaded); - - expect(assistant.content[0]).toMatchObject({ type: "thinking", thinking: "reasoning", thinkingSignature: "" }); - expect(assistant.content[1]).toMatchObject({ type: "text", text: "done", textSignature: "" }); - expect(assistant.content[2]).toMatchObject({ type: "toolCall", id: "tool_1", thoughtSignature: "" }); - }); - it("externalizes provider image data URLs and restores preserved history payloads across reload", async () => { using tempDir = TempDir.createSync("@pi-session-provider-image-persistence-"); const session = SessionManager.create(tempDir.path(), tempDir.path()); @@ -265,7 +229,8 @@ describe("SessionManager signature persistence", () => { it("drops a reasoning signature duplicated by the provider payload and keeps the payload on reload", async () => { using tempDir = TempDir.createSync("@pi-session-reasoning-dedup-e2e-"); const session = SessionManager.create(tempDir.path(), tempDir.path()); - const encrypted = "ENCRYPTED_REASONING_BLOB_UNIQUE_TOKEN"; + // >MAX_PERSIST_CHARS: regresses persistence truncating providerPayload reasoning items. + const encrypted = `ENCRYPTED_REASONING_BLOB_UNIQUE_TOKEN_${"E".repeat(600_000)}`; const reasoning = { type: "reasoning", id: "rs_1", encrypted_content: encrypted }; session.appendMessage({ role: "user", content: "continue", timestamp: 1 }); diff --git a/packages/collab-web/CHANGELOG.md b/packages/collab-web/CHANGELOG.md index 50eddd63d..790283c46 100644 --- a/packages/collab-web/CHANGELOG.md +++ b/packages/collab-web/CHANGELOG.md @@ -5,8 +5,8 @@ ### Fixed - Fixed missing response controls for "ask" questions in the mobile collaboration web UI. -- Fixed an issue where re-sending the same editor "ask" request would clear a guest's in-progress draft response. -- Fixed infinite retry loops in the agent transcript drawer when encountering terminal errors, ensuring the error is displayed and polling stops. +- Fixed an issue where re-sending an editor "ask" request would clear a guest's in-progress draft response. +- Fixed infinite retry loops in the agent transcript drawer by ensuring terminal errors are displayed and polling stops. - Fixed a delay in displaying pre-welcome connection errors (such as protocol version rejections), allowing the session to terminate immediately with the host's error reason. ## [16.2.0] - 2026-06-27 diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index 3c8fac117..8ef28c098 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -4,7 +4,7 @@ ### Changed -- Optimized stale-anchor remap validation from quadratic to linear complexity, significantly improving performance on large files. +- Significantly improved performance on large files by optimizing stale-anchor remap validation. ### Fixed diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index f0842b21a..33909473c 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added workingDir to ShellRunResult to allow hosts to synchronize the session's current working directory without executing a hidden probe command. +- Added `workingDir` to `ShellRunResult` to allow hosts to synchronize the session's current working directory without executing a hidden probe command. ### Fixed diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 0aac491f4..3c4b6bb1d 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed a potential event loop hang caused by processing oversized, unterminated terminal escape sequences (OSC/DCS/APC). +- Fixed a potential event loop hang when processing oversized, unterminated terminal escape sequences (OSC/DCS/APC). - Fixed an issue where large Windows terminal session restores could get truncated mid-frame during ConPTY full-paint resume. ## [16.2.13] - 2026-07-01 diff --git a/packages/wire/CHANGELOG.md b/packages/wire/CHANGELOG.md index 801445b38..a51f3b768 100644 --- a/packages/wire/CHANGELOG.md +++ b/packages/wire/CHANGELOG.md @@ -4,11 +4,11 @@ ### Breaking Changes -- Upgraded the collaboration protocol (COLLAB_PROTO) to version 3. Guests using version 2 are now rejected during the handshake with a protocol-mismatch error due to new interactive UI request/response requirements. +- Upgraded the collaboration protocol to version 3. Guests using version 2 will now be rejected during the handshake with a protocol-mismatch error. ### Added -- Added collaboration UI request and response frames, enabling browser guests to respond to interactive prompts initiated by the host. +- Added support for interactive UI request and response frames, enabling browser guests to respond to prompts initiated by the host. ## [16.1.8] - 2026-06-20