From 6256f97eaf32e74e340b27bb6ee8110d67e83316 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 20:05:04 +0000 Subject: [PATCH 01/88] fix(tui): aligned /extensions MCP status with /mcp list Two read paths previously diverged on whether an MCP server was active or disabled. /mcp list (slash-commands/helpers/mcp.ts:388) treats a server as disabled when config.enabled === false OR the name is in the user-level disabledServers denylist; the runtime MCP loader does the same in mcp/config.ts:115. The /extensions dashboard only consulted the dashboard-private settings.disabledExtensions array, so a server disabled via /mcp disable or enabled:false kept showing as active. Toggling MCP servers from the dashboard had the mirror problem: it only wrote to settings.disabledExtensions, so /mcp list never noticed. - state-manager: read user-level disabledServers from mcp.json once and consider enabled:false / denylist membership when deriving each MCP extension's state, matching /mcp list semantics. - extension-dashboard: route mcp:* toggles through setServerDisabled against the canonical mcp.json denylist, and clean any legacy settings.disabledExtensions entry on re-enable so it doesn't keep the server marked disabled. - Added a regression test exercising both read signals and the setServerDisabled round-trip the dashboard's MCP toggle now uses. Fixes #3827 --- packages/coding-agent/CHANGELOG.md | 4 + .../extensions/extension-dashboard.ts | 32 +++++ .../components/extensions/state-manager.ts | 13 +- .../extension-dashboard-mcp-parity.test.ts | 112 ++++++++++++++++++ 4 files changed, 158 insertions(+), 3 deletions(-) create mode 100644 packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..e33178c58 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`. The dashboard now mirrors `/mcp list` by honoring both the per-server `enabled: false` flag and the user-level `disabledServers` denylist, and toggling an MCP server from the dashboard now writes through the same denylist so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts index bc998b528..1fc88a1eb 100644 --- a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts +++ b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts @@ -24,7 +24,9 @@ import { truncateToWidth, visibleWidth, } from "@oh-my-pi/pi-tui"; +import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils"; import { Settings } from "../../../config/settings"; +import { setServerDisabled } from "../../../mcp/config-writer"; import { getTabBarTheme } from "../../../modes/shared"; import { theme } from "../../../modes/theme/theme"; import { matchesAppInterrupt } from "../../../modes/utils/keybinding-matchers"; @@ -263,6 +265,14 @@ export class ExtensionDashboard implements Component { const sm = this.settings ?? Settings.instance; if (!sm) return; + // MCP toggles route through the canonical denylist in + // `~/.omp/agent/mcp.json` so `/mcp list`, the MCP runtime, and this + // dashboard agree on every server's enabled state (issue #3827). + if (extensionId.startsWith("mcp:")) { + void this.#toggleMcpExtension(extensionId, enabled, sm); + return; + } + const disabled = ((sm.get("disabledExtensions") as string[]) ?? []).slice(); if (enabled) { const index = disabled.indexOf(extensionId); @@ -281,6 +291,28 @@ export class ExtensionDashboard implements Component { void this.#refreshFromState(); } + async #toggleMcpExtension(extensionId: string, enabled: boolean, sm: Settings): Promise { + const name = extensionId.slice("mcp:".length); + try { + await setServerDisabled(getMCPConfigPath("user", this.cwd), name, !enabled); + } catch (error) { + logger.warn("Failed to persist MCP toggle", { name, enabled, error: String(error) }); + } + + // Reconcile `settings.disabledExtensions` with the canonical denylist so + // a legacy `mcp:` flag from before this routing change doesn't keep + // the server marked disabled after the user re-enables it via the UI. + const stored = ((sm.get("disabledExtensions") as string[]) ?? []).slice(); + const had = stored.indexOf(extensionId); + if (enabled && had !== -1) { + stored.splice(had, 1); + sm.set("disabledExtensions", stored); + this.#applyDisabledExtensions(stored); + } + + await this.#refreshFromState(); + } + async #refreshFromState(): Promise { const refreshToken = ++this.#refreshToken; // Remember the current tab so it survives the re-sort. diff --git a/packages/coding-agent/src/modes/components/extensions/state-manager.ts b/packages/coding-agent/src/modes/components/extensions/state-manager.ts index 82adfadb0..82e42b1d0 100644 --- a/packages/coding-agent/src/modes/components/extensions/state-manager.ts +++ b/packages/coding-agent/src/modes/components/extensions/state-manager.ts @@ -4,7 +4,7 @@ */ import * as path from "node:path"; import { fuzzyMatch } from "@oh-my-pi/pi-tui"; -import { logger } from "@oh-my-pi/pi-utils"; +import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils"; import type { ContextFile } from "../../../capability/context-file"; import type { ExtensionModule } from "../../../capability/extension-module"; import type { Hook } from "../../../capability/hook"; @@ -22,6 +22,7 @@ import { isProviderEnabled, loadCapability, } from "../../../discovery"; +import { readDisabledServers } from "../../../mcp/config-writer"; import type { DashboardState, Extension, @@ -141,12 +142,18 @@ export async function loadAllExtensions(cwd?: string, disabledIds?: string[]): P logger.warn("Failed to load extension-modules capability", { error: String(error) }); } - // Load MCP servers + // Load MCP servers. The dashboard mirrors `/mcp list` (issue #3827) by + // honoring the same disable signals: the dashboard-private settings list, + // the per-server `enabled: false` flag, and the user-level `disabledServers` + // denylist that `/mcp disable` writes through `setServerDisabled`. try { + const mcpDisabledNames = cwd + ? new Set(await readDisabledServers(getMCPConfigPath("user", cwd)).catch(() => [])) + : new Set(); const mcps = await loadCapability("mcps", loadOpts); for (const server of mcps.all) { const id = makeExtensionId("mcp", server.name); - const isDisabled = disabledExtensions.has(id); + const isDisabled = disabledExtensions.has(id) || server.enabled === false || mcpDisabledNames.has(server.name); const isShadowed = (server as { _shadowed?: boolean })._shadowed; const providerEnabled = isProviderEnabled(server._source.provider); diff --git a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts new file mode 100644 index 000000000..3981ca570 --- /dev/null +++ b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts @@ -0,0 +1,112 @@ +/** + * Regression guard for issue #3827. + * + * `/mcp list` and the `/extensions` dashboard MUST agree on whether a given MCP + * server is enabled or disabled. The two read paths historically diverged: the + * dashboard's `loadAllExtensions` only consulted the dashboard-private + * `disabledExtensions` settings array, while `/mcp list` (and the MCP runtime + * itself) honored both the per-server `enabled` flag in `mcp.json` and the + * user-level `disabledServers` denylist. + * + * The fixtures below cover both inputs and the round-trip the dashboard's + * MCP toggle uses (`setServerDisabled`). + */ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { initializeWithSettings } from "@oh-my-pi/pi-coding-agent/discovery"; +import { setServerDisabled } from "@oh-my-pi/pi-coding-agent/mcp/config-writer"; +import { loadAllExtensions } from "@oh-my-pi/pi-coding-agent/modes/components/extensions/state-manager"; +import { __resetDirsFromEnvForTests, getMCPConfigPath, removeWithRetries, setAgentDir } from "@oh-my-pi/pi-utils"; + +describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { + let projectDir = ""; + let userAgentDir = ""; + + beforeEach(async () => { + resetSettingsForTest(); + projectDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-3827-project-")); + userAgentDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-3827-user-")); + + // Redirect user-scoped mcp.json (resolved via getAgentDir() at the call + // site) into the per-test temp directory so neither the discovery loader + // nor the denylist reader touches the real user profile. + setAgentDir(userAgentDir); + + await fs.mkdir(path.join(projectDir, ".omp"), { recursive: true }); + await fs.writeFile( + path.join(projectDir, ".omp", "mcp.json"), + JSON.stringify({ + mcpServers: { + "denylisted-server": { command: "echo", args: ["denylisted"] }, + "flag-disabled-server": { command: "echo", args: ["flag"], enabled: false }, + "active-server": { command: "echo", args: ["active"] }, + }, + }), + ); + + // User-level mcp.json carries the denylist; this is what `/mcp disable` + // writes through setServerDisabled(). + await fs.writeFile( + path.join(userAgentDir, "mcp.json"), + JSON.stringify({ + mcpServers: {}, + disabledServers: ["denylisted-server"], + }), + ); + + const settings = await Settings.init({ inMemory: true, cwd: projectDir }); + initializeWithSettings(settings); + }); + + afterEach(async () => { + resetSettingsForTest(); + __resetDirsFromEnvForTests(); + await removeWithRetries(projectDir); + await removeWithRetries(userAgentDir); + }); + + test("treats a server in user-level disabledServers as disabled (matches /mcp list)", async () => { + const extensions = await loadAllExtensions(projectDir, []); + const denylisted = extensions.find(e => e.id === "mcp:denylisted-server"); + expect(denylisted).toBeDefined(); + expect(denylisted!.state).toBe("disabled"); + expect(denylisted!.disabledReason).toBe("item-disabled"); + }); + + test("treats a server with enabled:false as disabled (matches /mcp list)", async () => { + const extensions = await loadAllExtensions(projectDir, []); + const flagDisabled = extensions.find(e => e.id === "mcp:flag-disabled-server"); + expect(flagDisabled).toBeDefined(); + expect(flagDisabled!.state).toBe("disabled"); + expect(flagDisabled!.disabledReason).toBe("item-disabled"); + }); + + test("leaves untouched servers active", async () => { + const extensions = await loadAllExtensions(projectDir, []); + const active = extensions.find(e => e.id === "mcp:active-server"); + expect(active).toBeDefined(); + expect(active!.state).toBe("active"); + expect(active!.disabledReason).toBeUndefined(); + }); + + test("setServerDisabled round-trips through the dashboard view", async () => { + // Re-enable `denylisted-server` through the canonical writer the + // dashboard's MCP toggle now calls. The dashboard view MUST flip to + // active on the next load. + await setServerDisabled(getMCPConfigPath("user", projectDir), "denylisted-server", false); + const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:denylisted-server"); + expect(reenabled).toBeDefined(); + expect(reenabled!.state).toBe("active"); + + // The inverse path: disabling `active-server` via the writer flips the + // dashboard view to disabled. + await setServerDisabled(getMCPConfigPath("user", projectDir), "active-server", true); + const disabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:active-server"); + expect(disabled).toBeDefined(); + expect(disabled!.state).toBe("disabled"); + expect(disabled!.disabledReason).toBe("item-disabled"); + }); +}); From 812b246e7e663a85fdde4f41c6e732ef55bab985 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 20:13:15 +0000 Subject: [PATCH 02/88] fix(tui): dashboard re-enable flips enabled:false in mcp.json Codex review on #3829: when an MCP server's mcp.json entry carries enabled:false, the dashboard toggle previously only removed the name from the user-level disabledServers denylist. state-manager's new `server.enabled === false` check (state-manager.ts:156) then still marked the row disabled, leaving such servers impossible to re-enable from /extensions. Extracted setMcpServerEnabled() into mcp/config-writer.ts mirroring /mcp enable | /mcp disable semantics: - Server defined in project mcp.json -> update enabled on that entry. - Else server defined in user mcp.json -> update enabled on that entry. - Else (discovered third-party server) -> use the user-level disabledServers denylist. - On re-enable, always clear any stale denylist entry. extension-dashboard.ts routes mcp:* toggles through this helper. Added four new regression tests covering: enabled:false re-enable, mixed flag+denylist re-enable, disable on a config-resident server writing enabled:false (not denylist), and discovered-server denylist round-trip. Fixes #3827 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/mcp/config-writer.ts | 48 +++++++++ .../extensions/extension-dashboard.ts | 16 ++- .../extension-dashboard-mcp-parity.test.ts | 100 +++++++++++++++++- 4 files changed, 159 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e33178c58..5ed48f823 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`. The dashboard now mirrors `/mcp list` by honoring both the per-server `enabled: false` flag and the user-level `disabledServers` denylist, and toggling an MCP server from the dashboard now writes through the same denylist so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). +- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`. The dashboard now mirrors `/mcp list` by honoring both the per-server `enabled: false` flag and the user-level `disabledServers` denylist, and toggling an MCP server from the dashboard now writes through the same canonical mcp.json — flipping `enabled` on config-resident servers (so re-enabling a server with `enabled: false` actually re-enables it) and the denylist for discovered third-party servers, so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/src/mcp/config-writer.ts b/packages/coding-agent/src/mcp/config-writer.ts index 650a23687..28bf811c8 100644 --- a/packages/coding-agent/src/mcp/config-writer.ts +++ b/packages/coding-agent/src/mcp/config-writer.ts @@ -227,3 +227,51 @@ export async function setServerDisabled(filePath: string, name: string, disabled await writeMCPConfigFile(filePath, updated); } + +/** + * Flip a server's enabled/disabled state regardless of where it lives. + * + * Mirrors `/mcp enable` / `/mcp disable` (see + * `slash-commands/helpers/mcp.ts:handleEnableDisableCommand`) so the + * `/extensions` dashboard's MCP toggle and the slash command share one source + * of truth: + * + * - Server defined in project mcp.json → write `enabled` on that entry. + * - Else server defined in user mcp.json → write `enabled` on that entry. + * - Else (discovered third-party server) → use the user-level + * `disabledServers` denylist. + * - On re-enable, ALWAYS clear any stale denylist entry so a server disabled + * via `/mcp disable` and later re-enabled via `enabled: true` doesn't stay + * suppressed by a leftover deny entry. + */ +export async function setMcpServerEnabled( + userPath: string, + projectPath: string, + name: string, + enabled: boolean, +): Promise { + const [userConfig, projectConfig] = await Promise.all([readMCPConfigFile(userPath), readMCPConfigFile(projectPath)]); + + let updatedInConfig = false; + if (projectConfig.mcpServers?.[name] !== undefined) { + await updateMCPServer(projectPath, name, { ...projectConfig.mcpServers[name], enabled }); + updatedInConfig = true; + } else if (userConfig.mcpServers?.[name] !== undefined) { + await updateMCPServer(userPath, name, { ...userConfig.mcpServers[name], enabled }); + updatedInConfig = true; + } + + if (!updatedInConfig) { + await setServerDisabled(userPath, name, !enabled); + return; + } + + // Re-enable: the per-server `enabled` flag is now true, but a stale denylist + // entry would still hide the server. Drop it. + if (enabled) { + const denied = await readDisabledServers(userPath); + if (denied.includes(name)) { + await setServerDisabled(userPath, name, false); + } + } +} diff --git a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts index 1fc88a1eb..0afca59d5 100644 --- a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts +++ b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts @@ -26,7 +26,7 @@ import { } from "@oh-my-pi/pi-tui"; import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils"; import { Settings } from "../../../config/settings"; -import { setServerDisabled } from "../../../mcp/config-writer"; +import { setMcpServerEnabled } from "../../../mcp/config-writer"; import { getTabBarTheme } from "../../../modes/shared"; import { theme } from "../../../modes/theme/theme"; import { matchesAppInterrupt } from "../../../modes/utils/keybinding-matchers"; @@ -294,14 +294,20 @@ export class ExtensionDashboard implements Component { async #toggleMcpExtension(extensionId: string, enabled: boolean, sm: Settings): Promise { const name = extensionId.slice("mcp:".length); try { - await setServerDisabled(getMCPConfigPath("user", this.cwd), name, !enabled); + await setMcpServerEnabled( + getMCPConfigPath("user", this.cwd), + getMCPConfigPath("project", this.cwd), + name, + enabled, + ); } catch (error) { logger.warn("Failed to persist MCP toggle", { name, enabled, error: String(error) }); } - // Reconcile `settings.disabledExtensions` with the canonical denylist so - // a legacy `mcp:` flag from before this routing change doesn't keep - // the server marked disabled after the user re-enables it via the UI. + // Reconcile `settings.disabledExtensions` with the canonical mcp.json + // state so a legacy `mcp:` flag from before this routing change + // doesn't keep the server marked disabled after the user re-enables it + // via the UI. const stored = ((sm.get("disabledExtensions") as string[]) ?? []).slice(); const had = stored.indexOf(extensionId); if (enabled && had !== -1) { diff --git a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts index 3981ca570..1680bfb07 100644 --- a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts +++ b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts @@ -17,7 +17,7 @@ import * as os from "node:os"; import * as path from "node:path"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { initializeWithSettings } from "@oh-my-pi/pi-coding-agent/discovery"; -import { setServerDisabled } from "@oh-my-pi/pi-coding-agent/mcp/config-writer"; +import { readMCPConfigFile, setMcpServerEnabled, setServerDisabled } from "@oh-my-pi/pi-coding-agent/mcp/config-writer"; import { loadAllExtensions } from "@oh-my-pi/pi-coding-agent/modes/components/extensions/state-manager"; import { __resetDirsFromEnvForTests, getMCPConfigPath, removeWithRetries, setAgentDir } from "@oh-my-pi/pi-utils"; @@ -109,4 +109,102 @@ describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { expect(disabled!.state).toBe("disabled"); expect(disabled!.disabledReason).toBe("item-disabled"); }); + + test("dashboard re-enable flips enabled:false in mcp.json (PR #3829 review)", async () => { + // The bug: when a server has `enabled: false` in mcp.json, the dashboard + // toggle previously only removed it from the user-level denylist, so + // state-manager's `server.enabled === false` check kept it disabled. + // setMcpServerEnabled MUST overwrite the per-server flag. + const projectMcpPath = path.join(projectDir, ".omp", "mcp.json"); + + await setMcpServerEnabled( + getMCPConfigPath("user", projectDir), + getMCPConfigPath("project", projectDir), + "flag-disabled-server", + true, + ); + + const projectConfig = await readMCPConfigFile(projectMcpPath); + expect(projectConfig.mcpServers?.["flag-disabled-server"]?.enabled).toBe(true); + + const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:flag-disabled-server"); + expect(reenabled).toBeDefined(); + expect(reenabled!.state).toBe("active"); + }); + + test("dashboard re-enable also clears a stale denylist entry on a config-resident server", async () => { + // Manually disable `active-server` via BOTH the per-server flag and the + // denylist, simulating a server that's been toggled off multiple ways. + const projectMcpPath = path.join(projectDir, ".omp", "mcp.json"); + const initial = await readMCPConfigFile(projectMcpPath); + await Bun.write( + projectMcpPath, + JSON.stringify({ + ...initial, + mcpServers: { + ...initial.mcpServers, + "active-server": { ...initial.mcpServers!["active-server"], enabled: false }, + }, + }), + ); + await setServerDisabled(getMCPConfigPath("user", projectDir), "active-server", true); + + await setMcpServerEnabled( + getMCPConfigPath("user", projectDir), + getMCPConfigPath("project", projectDir), + "active-server", + true, + ); + + const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfig.disabledServers ?? []).not.toContain("active-server"); + + const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:active-server"); + expect(reenabled).toBeDefined(); + expect(reenabled!.state).toBe("active"); + }); + + test("dashboard disable on a config-resident server writes enabled:false (not denylist)", async () => { + await setMcpServerEnabled( + getMCPConfigPath("user", projectDir), + getMCPConfigPath("project", projectDir), + "active-server", + false, + ); + + const projectConfig = await readMCPConfigFile(path.join(projectDir, ".omp", "mcp.json")); + expect(projectConfig.mcpServers?.["active-server"]?.enabled).toBe(false); + + // The denylist is reserved for discovered (config-less) servers; a + // config-resident server's `enabled: false` flag is the canonical signal. + const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfig.disabledServers ?? []).not.toContain("active-server"); + + const disabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:active-server"); + expect(disabled).toBeDefined(); + expect(disabled!.state).toBe("disabled"); + }); + + test("dashboard toggles on a discovered (config-less) server use the denylist", async () => { + // `phantom-server` is not in any config; only the denylist can suppress it. + await setMcpServerEnabled( + getMCPConfigPath("user", projectDir), + getMCPConfigPath("project", projectDir), + "phantom-server", + false, + ); + + let userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfig.disabledServers ?? []).toContain("phantom-server"); + + await setMcpServerEnabled( + getMCPConfigPath("user", projectDir), + getMCPConfigPath("project", projectDir), + "phantom-server", + true, + ); + + userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfig.disabledServers ?? []).not.toContain("phantom-server"); + }); }); From 16ef3c54f43c74a9a53d5e1dfec44ca0b5bf5650 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 20:26:16 +0000 Subject: [PATCH 03/88] fix(tui): re-enabled MCP alternate config sources Codex review on #3829: the dashboard re-enable path still missed MCP servers loaded from supported non-primary native config files such as .omp/.mcp.json or user .mcp.json. Those rows carry enabled:false from their source file, so falling back to the user disabledServers denylist could not make the row active again. - setMcpServerEnabled now accepts the loaded row's sourcePath and checks it before the primary project/user mcp.json paths. - extension-dashboard passes the source path for writable MCP providers (native and mcp-json), avoiding accidental edits to third-party tool configs while still updating .omp/.mcp.json and standalone MCP JSON sources. - Added a regression test for a server loaded from .omp/.mcp.json with enabled:false; re-enable flips that file to enabled:true and does not write the denylist. Fixes #3827 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/mcp/config-writer.ts | 44 +++++---- .../extensions/extension-dashboard.ts | 16 ++- .../extension-dashboard-mcp-parity.test.ts | 97 +++++++++++++------ 4 files changed, 104 insertions(+), 55 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5ed48f823..7d4309e1d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`. The dashboard now mirrors `/mcp list` by honoring both the per-server `enabled: false` flag and the user-level `disabledServers` denylist, and toggling an MCP server from the dashboard now writes through the same canonical mcp.json — flipping `enabled` on config-resident servers (so re-enabling a server with `enabled: false` actually re-enables it) and the denylist for discovered third-party servers, so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). +- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`. The dashboard now mirrors `/mcp list` by honoring both the per-server `enabled: false` flag and the user-level `disabledServers` denylist, and toggling an MCP server from the dashboard now writes through the same canonical mcp.json path — flipping `enabled` on the loaded source file for config-resident servers, including supported non-primary files such as `.omp/.mcp.json`, and the denylist for discovered third-party servers, so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/src/mcp/config-writer.ts b/packages/coding-agent/src/mcp/config-writer.ts index 28bf811c8..ef41e8158 100644 --- a/packages/coding-agent/src/mcp/config-writer.ts +++ b/packages/coding-agent/src/mcp/config-writer.ts @@ -228,15 +228,25 @@ export async function setServerDisabled(filePath: string, name: string, disabled await writeMCPConfigFile(filePath, updated); } +/** Paths and target state for toggling one MCP server across known config files. */ +export interface SetMcpServerEnabledOptions { + userPath: string; + projectPath: string; + sourcePath?: string; + name: string; + enabled: boolean; +} + /** * Flip a server's enabled/disabled state regardless of where it lives. * - * Mirrors `/mcp enable` / `/mcp disable` (see - * `slash-commands/helpers/mcp.ts:handleEnableDisableCommand`) so the - * `/extensions` dashboard's MCP toggle and the slash command share one source - * of truth: + * Mirrors `/mcp enable` / `/mcp disable` for primary configs, but also accepts + * the loaded row's source path so dashboard toggles can update supported + * alternate MCP files such as `.omp/.mcp.json` and user `.mcp.json` before + * falling back to the user-level `disabledServers` denylist. * - * - Server defined in project mcp.json → write `enabled` on that entry. + * - Server found in `sourcePath` → write `enabled` on that entry. + * - Else server defined in project mcp.json → write `enabled` on that entry. * - Else server defined in user mcp.json → write `enabled` on that entry. * - Else (discovered third-party server) → use the user-level * `disabledServers` denylist. @@ -244,21 +254,19 @@ export async function setServerDisabled(filePath: string, name: string, disabled * via `/mcp disable` and later re-enabled via `enabled: true` doesn't stay * suppressed by a leftover deny entry. */ -export async function setMcpServerEnabled( - userPath: string, - projectPath: string, - name: string, - enabled: boolean, -): Promise { - const [userConfig, projectConfig] = await Promise.all([readMCPConfigFile(userPath), readMCPConfigFile(projectPath)]); - +export async function setMcpServerEnabled(options: SetMcpServerEnabledOptions): Promise { + const { userPath, projectPath, sourcePath, name, enabled } = options; + const candidatePaths = [...new Set([sourcePath, projectPath, userPath].filter(path => path !== undefined))]; let updatedInConfig = false; - if (projectConfig.mcpServers?.[name] !== undefined) { - await updateMCPServer(projectPath, name, { ...projectConfig.mcpServers[name], enabled }); - updatedInConfig = true; - } else if (userConfig.mcpServers?.[name] !== undefined) { - await updateMCPServer(userPath, name, { ...userConfig.mcpServers[name], enabled }); + + for (const filePath of candidatePaths) { + const config = await readMCPConfigFile(filePath); + const server = config.mcpServers?.[name]; + if (server === undefined) continue; + + await updateMCPServer(filePath, name, { ...server, enabled }); updatedInConfig = true; + break; } if (!updatedInConfig) { diff --git a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts index 0afca59d5..028eda42f 100644 --- a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts +++ b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts @@ -294,12 +294,13 @@ export class ExtensionDashboard implements Component { async #toggleMcpExtension(extensionId: string, enabled: boolean, sm: Settings): Promise { const name = extensionId.slice("mcp:".length); try { - await setMcpServerEnabled( - getMCPConfigPath("user", this.cwd), - getMCPConfigPath("project", this.cwd), + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", this.cwd), + projectPath: getMCPConfigPath("project", this.cwd), + sourcePath: this.#writableMcpSourcePath(extensionId), name, enabled, - ); + }); } catch (error) { logger.warn("Failed to persist MCP toggle", { name, enabled, error: String(error) }); } @@ -319,6 +320,13 @@ export class ExtensionDashboard implements Component { await this.#refreshFromState(); } + #writableMcpSourcePath(extensionId: string): string | undefined { + const extension = this.#state.extensions.find(ext => ext.id === extensionId); + if (!extension) return undefined; + if (extension.source.provider !== "native" && extension.source.provider !== "mcp-json") return undefined; + return extension.path; + } + async #refreshFromState(): Promise { const refreshToken = ++this.#refreshToken; // Remember the current tab so it survives the re-sort. diff --git a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts index 1680bfb07..84c822a5f 100644 --- a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts +++ b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts @@ -8,8 +8,8 @@ * itself) honored both the per-server `enabled` flag in `mcp.json` and the * user-level `disabledServers` denylist. * - * The fixtures below cover both inputs and the round-trip the dashboard's - * MCP toggle uses (`setServerDisabled`). + * The fixtures below cover both inputs and the round-trip helper the + * dashboard's MCP toggle uses. */ import { afterEach, beforeEach, describe, expect, test } from "bun:test"; import * as fs from "node:fs/promises"; @@ -117,12 +117,12 @@ describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { // setMcpServerEnabled MUST overwrite the per-server flag. const projectMcpPath = path.join(projectDir, ".omp", "mcp.json"); - await setMcpServerEnabled( - getMCPConfigPath("user", projectDir), - getMCPConfigPath("project", projectDir), - "flag-disabled-server", - true, - ); + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "flag-disabled-server", + enabled: true, + }); const projectConfig = await readMCPConfigFile(projectMcpPath); expect(projectConfig.mcpServers?.["flag-disabled-server"]?.enabled).toBe(true); @@ -149,12 +149,12 @@ describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { ); await setServerDisabled(getMCPConfigPath("user", projectDir), "active-server", true); - await setMcpServerEnabled( - getMCPConfigPath("user", projectDir), - getMCPConfigPath("project", projectDir), - "active-server", - true, - ); + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "active-server", + enabled: true, + }); const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); expect(userConfig.disabledServers ?? []).not.toContain("active-server"); @@ -165,12 +165,12 @@ describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { }); test("dashboard disable on a config-resident server writes enabled:false (not denylist)", async () => { - await setMcpServerEnabled( - getMCPConfigPath("user", projectDir), - getMCPConfigPath("project", projectDir), - "active-server", - false, - ); + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "active-server", + enabled: false, + }); const projectConfig = await readMCPConfigFile(path.join(projectDir, ".omp", "mcp.json")); expect(projectConfig.mcpServers?.["active-server"]?.enabled).toBe(false); @@ -185,24 +185,57 @@ describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { expect(disabled!.state).toBe("disabled"); }); + test("dashboard re-enable updates the row's non-primary source mcp.json before denylisting", async () => { + const alternatePath = path.join(projectDir, ".omp", ".mcp.json"); + await Bun.write( + alternatePath, + JSON.stringify({ + mcpServers: { + "alternate-server": { command: "echo", args: ["alternate"], enabled: false }, + }, + }), + ); + + const disabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:alternate-server"); + expect(disabled).toBeDefined(); + expect(disabled!.state).toBe("disabled"); + + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + sourcePath: alternatePath, + name: "alternate-server", + enabled: true, + }); + + const alternateConfig = await readMCPConfigFile(alternatePath); + expect(alternateConfig.mcpServers?.["alternate-server"]?.enabled).toBe(true); + + const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfig.disabledServers ?? []).not.toContain("alternate-server"); + + const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:alternate-server"); + expect(reenabled).toBeDefined(); + expect(reenabled!.state).toBe("active"); + }); test("dashboard toggles on a discovered (config-less) server use the denylist", async () => { // `phantom-server` is not in any config; only the denylist can suppress it. - await setMcpServerEnabled( - getMCPConfigPath("user", projectDir), - getMCPConfigPath("project", projectDir), - "phantom-server", - false, - ); + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "phantom-server", + enabled: false, + }); let userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); expect(userConfig.disabledServers ?? []).toContain("phantom-server"); - await setMcpServerEnabled( - getMCPConfigPath("user", projectDir), - getMCPConfigPath("project", projectDir), - "phantom-server", - true, - ); + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "phantom-server", + enabled: true, + }); userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); expect(userConfig.disabledServers ?? []).not.toContain("phantom-server"); From e34f2a81e9d00279b0420a0b0dce881acf1f9f6a Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 20:39:50 +0000 Subject: [PATCH 04/88] fix(tui): force-enabled MCP from tool-owned sources via enabledServers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex review on #3829: when an MCP server lives in a non-writable source config such as opencode.json with enabled:false, the dashboard re-enable had nowhere to write to — the writable mcp.json fallback did not own the server, so setMcpServerEnabled fell through to the denylist and the source's enabled:false kept the row disabled. Added a parallel allowlist to the user-level mcp.json that overrides a non-writable source's enabled:false flag without ever mutating the foreign config: - types + schema: new enabledServers array (mirrors disabledServers). - config-writer: readEnabledServers + setServerForceEnabled helpers, and setMcpServerEnabled now writes to enabledServers on enable when no writable mcp.json owns the server, clears it whenever a writable source becomes the source of truth, and always clears the override on disable so a force-enabled server can be turned off. - mcp/config (runtime loader) and state-manager (dashboard read): honor enabledServers as an override on enabled:false, while still letting disabledServers win. - Added a regression test that walks the full lifecycle for an opencode.json server: enabled:false is surfaced as disabled, the dashboard re-enable force-enables via enabledServers without touching opencode.json, then disable clears the override and populates disabledServers. Fixes #3827 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/config/mcp-schema.json | 11 +- .../coding-agent/src/mcp/config-writer.ts | 103 ++++++++++++++---- packages/coding-agent/src/mcp/config.ts | 16 ++- packages/coding-agent/src/mcp/types.ts | 3 + .../components/extensions/state-manager.ts | 26 ++++- .../extension-dashboard-mcp-parity.test.ts | 69 +++++++++++- 7 files changed, 196 insertions(+), 34 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7d4309e1d..f119eb07d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`. The dashboard now mirrors `/mcp list` by honoring both the per-server `enabled: false` flag and the user-level `disabledServers` denylist, and toggling an MCP server from the dashboard now writes through the same canonical mcp.json path — flipping `enabled` on the loaded source file for config-resident servers, including supported non-primary files such as `.omp/.mcp.json`, and the denylist for discovered third-party servers, so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). +- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`, and made `/extensions` re-enable work for every supported source. The dashboard now mirrors `/mcp list` by honoring the per-server `enabled: false` flag plus the user-level `disabledServers` denylist and a new `enabledServers` allowlist; the dashboard's MCP toggle writes through the canonical mcp.json — flipping `enabled` on the loaded source file for config-resident servers (including supported non-primary files such as `.omp/.mcp.json`), force-enabling tool-owned sources (such as `opencode.json`) via the user `enabledServers` allowlist without mutating the foreign config, and using the `disabledServers` denylist for purely discovered third-party servers, so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/src/config/mcp-schema.json b/packages/coding-agent/src/config/mcp-schema.json index 33dc7873f..ce0cbbcca 100644 --- a/packages/coding-agent/src/config/mcp-schema.json +++ b/packages/coding-agent/src/config/mcp-schema.json @@ -22,7 +22,16 @@ }, "disabledServers": { "type": "array", - "description": "User-level denylist for disabling discovered servers by name.", + "description": "User-level denylist for disabling discovered servers by name. Highest precedence: a server here is hidden regardless of any other source.", + "items": { + "type": "string", + "minLength": 1 + }, + "uniqueItems": true + }, + "enabledServers": { + "type": "array", + "description": "User-level allowlist that overrides a discovered server's `enabled: false` flag (e.g. when the source config is owned by another tool such as opencode.json). The denylist still wins.", "items": { "type": "string", "minLength": 1 diff --git a/packages/coding-agent/src/mcp/config-writer.ts b/packages/coding-agent/src/mcp/config-writer.ts index ef41e8158..238f13e91 100644 --- a/packages/coding-agent/src/mcp/config-writer.ts +++ b/packages/coding-agent/src/mcp/config-writer.ts @@ -228,10 +228,52 @@ export async function setServerDisabled(filePath: string, name: string, disabled await writeMCPConfigFile(filePath, updated); } +/** + * Read the user-level force-enable list (allowlist that overrides a + * non-writable source config's `enabled: false`). + */ +export async function readEnabledServers(filePath: string): Promise { + const config = await readMCPConfigFile(filePath); + return Array.isArray(config.enabledServers) ? config.enabledServers : []; +} + +/** + * Add or remove a server name from the user-level force-enable list. + * The list overrides a discovered server's `enabled: false` flag but does + * NOT override the `disabledServers` denylist. + */ +export async function setServerForceEnabled(filePath: string, name: string, force: boolean): Promise { + const config = await readMCPConfigFile(filePath); + const current = new Set(config.enabledServers ?? []); + + if (force) { + current.add(name); + } else { + current.delete(name); + } + + const updated: MCPConfigFile = { + ...config, + enabledServers: current.size > 0 ? Array.from(current).sort() : undefined, + }; + + if (!updated.enabledServers) { + delete updated.enabledServers; + } + + await writeMCPConfigFile(filePath, updated); +} + /** Paths and target state for toggling one MCP server across known config files. */ export interface SetMcpServerEnabledOptions { userPath: string; projectPath: string; + /** + * Absolute path to the loaded row's source mcp.json. Provide ONLY for + * formats this codebase owns (native `.omp/mcp.json` and `mcp-json` + * `mcp.json`/`.mcp.json`). Tool-owned configs (opencode.json, claude.json, + * settings.json …) MUST be omitted; we never mutate another tool's file. + */ sourcePath?: string; name: string; enabled: boolean; @@ -240,19 +282,24 @@ export interface SetMcpServerEnabledOptions { /** * Flip a server's enabled/disabled state regardless of where it lives. * - * Mirrors `/mcp enable` / `/mcp disable` for primary configs, but also accepts - * the loaded row's source path so dashboard toggles can update supported - * alternate MCP files such as `.omp/.mcp.json` and user `.mcp.json` before - * falling back to the user-level `disabledServers` denylist. + * Resolution order, mirroring `/mcp enable` / `/mcp disable` plus the dashboard + * fix for non-writable source configs: * - * - Server found in `sourcePath` → write `enabled` on that entry. - * - Else server defined in project mcp.json → write `enabled` on that entry. - * - Else server defined in user mcp.json → write `enabled` on that entry. - * - Else (discovered third-party server) → use the user-level - * `disabledServers` denylist. - * - On re-enable, ALWAYS clear any stale denylist entry so a server disabled - * via `/mcp disable` and later re-enabled via `enabled: true` doesn't stay - * suppressed by a leftover deny entry. + * - Server found in `sourcePath` (writable) → write `enabled` on that entry. + * - Else server in project mcp.json → write `enabled` there. + * - Else server in user mcp.json → write `enabled` there. + * - Else (server defined in a tool-owned source like opencode.json, OR a + * purely discovered server): + * - Disable → add to the user-level `disabledServers` denylist. + * - Enable → add to the user-level `enabledServers` allowlist so the + * dashboard / runtime override the non-writable source's + * `enabled: false` flag. + * + * Cleanup invariants — on every call: + * - Re-enable clears any stale denylist entry so a server disabled via + * `/mcp disable` and re-enabled here doesn't stay suppressed. + * - Disable clears any stale allowlist entry so re-disabling a + * force-enabled server actually takes effect. */ export async function setMcpServerEnabled(options: SetMcpServerEnabledOptions): Promise { const { userPath, projectPath, sourcePath, name, enabled } = options; @@ -269,17 +316,35 @@ export async function setMcpServerEnabled(options: SetMcpServerEnabledOptions): break; } - if (!updatedInConfig) { - await setServerDisabled(userPath, name, !enabled); - return; - } - - // Re-enable: the per-server `enabled` flag is now true, but a stale denylist - // entry would still hide the server. Drop it. if (enabled) { + // Either we just wrote `enabled: true` on a writable source, or the + // server lives in a non-writable source whose `enabled: false` flag we + // need to override via the user allowlist. Either way the denylist + // entry (if any) must clear so the row becomes active. const denied = await readDisabledServers(userPath); if (denied.includes(name)) { await setServerDisabled(userPath, name, false); } + + const forced = await readEnabledServers(userPath); + const isForced = forced.includes(name); + if (!updatedInConfig && !isForced) { + await setServerForceEnabled(userPath, name, true); + } else if (updatedInConfig && isForced) { + // Writable source now carries `enabled: true`; the override is + // redundant. Drop it so the user's allowlist stays tidy. + await setServerForceEnabled(userPath, name, false); + } + return; + } + + // Disable path. Clear any force-enable override regardless of source so the + // disable actually sticks. + const forced = await readEnabledServers(userPath); + if (forced.includes(name)) { + await setServerForceEnabled(userPath, name, false); + } + if (!updatedInConfig) { + await setServerDisabled(userPath, name, true); } } diff --git a/packages/coding-agent/src/mcp/config.ts b/packages/coding-agent/src/mcp/config.ts index c3fff09b6..39872005b 100644 --- a/packages/coding-agent/src/mcp/config.ts +++ b/packages/coding-agent/src/mcp/config.ts @@ -9,7 +9,7 @@ import { mcpCapability } from "../capability/mcp"; import type { SourceMeta } from "../capability/types"; import type { MCPServer } from "../discovery"; import { loadCapability } from "../discovery"; -import { readDisabledServers } from "./config-writer"; +import { readDisabledServers, readEnabledServers } from "./config-writer"; import type { MCPServerConfig } from "./types"; /** Options for loading MCP configs */ @@ -105,16 +105,20 @@ export async function loadAllMCPConfigs(cwd: string, options?: LoadMCPConfigsOpt ? result.items : result.items.filter(server => server._source.level !== "project"); - // Load user-level disabled servers list - const disabledServers = new Set(await readDisabledServers(getMCPConfigPath("user", cwd))); + // Load user-level disable/force-enable lists. The denylist always wins; the + // allowlist overrides a non-writable source config's `enabled: false`. + const userPath = getMCPConfigPath("user", cwd); + const [disabledServers, forcedEnabled] = await Promise.all([ + readDisabledServers(userPath).then(list => new Set(list)), + readEnabledServers(userPath).then(list => new Set(list)), + ]); // Convert to legacy format and preserve source metadata let configs: Record = {}; let sources: Record = {}; for (const server of servers) { const config = convertToLegacyConfig(server); - if (config.enabled === false || disabledServers.has(server.name)) { - continue; - } + if (disabledServers.has(server.name)) continue; + if (config.enabled === false && !forcedEnabled.has(server.name)) continue; configs[server.name] = config; sources[server.name] = server._source; } diff --git a/packages/coding-agent/src/mcp/types.ts b/packages/coding-agent/src/mcp/types.ts index ae9af5039..c526cf8c5 100644 --- a/packages/coding-agent/src/mcp/types.ts +++ b/packages/coding-agent/src/mcp/types.ts @@ -111,7 +111,10 @@ export const MCP_CONFIG_SCHEMA_URL = export interface MCPConfigFile { $schema?: string; mcpServers?: Record; + /** Names to hide regardless of any source `enabled` flag. Highest precedence. */ disabledServers?: string[]; + /** Names to force-enable when a non-writable source reports `enabled: false`. */ + enabledServers?: string[]; } // ============================================================================= diff --git a/packages/coding-agent/src/modes/components/extensions/state-manager.ts b/packages/coding-agent/src/modes/components/extensions/state-manager.ts index 82e42b1d0..e368a1f11 100644 --- a/packages/coding-agent/src/modes/components/extensions/state-manager.ts +++ b/packages/coding-agent/src/modes/components/extensions/state-manager.ts @@ -22,7 +22,7 @@ import { isProviderEnabled, loadCapability, } from "../../../discovery"; -import { readDisabledServers } from "../../../mcp/config-writer"; +import { readDisabledServers, readEnabledServers } from "../../../mcp/config-writer"; import type { DashboardState, Extension, @@ -145,15 +145,29 @@ export async function loadAllExtensions(cwd?: string, disabledIds?: string[]): P // Load MCP servers. The dashboard mirrors `/mcp list` (issue #3827) by // honoring the same disable signals: the dashboard-private settings list, // the per-server `enabled: false` flag, and the user-level `disabledServers` - // denylist that `/mcp disable` writes through `setServerDisabled`. + // denylist that `/mcp disable` writes through `setServerDisabled`. The + // user-level `enabledServers` allowlist overrides a non-writable source's + // `enabled: false` (e.g. opencode.json) but never the denylist. try { - const mcpDisabledNames = cwd - ? new Set(await readDisabledServers(getMCPConfigPath("user", cwd)).catch(() => [])) - : new Set(); + const userMcpPath = cwd ? getMCPConfigPath("user", cwd) : undefined; + const [mcpDisabledNames, mcpForcedEnabled] = await Promise.all([ + userMcpPath + ? readDisabledServers(userMcpPath) + .then(list => new Set(list)) + .catch(() => new Set()) + : Promise.resolve(new Set()), + userMcpPath + ? readEnabledServers(userMcpPath) + .then(list => new Set(list)) + .catch(() => new Set()) + : Promise.resolve(new Set()), + ]); const mcps = await loadCapability("mcps", loadOpts); for (const server of mcps.all) { const id = makeExtensionId("mcp", server.name); - const isDisabled = disabledExtensions.has(id) || server.enabled === false || mcpDisabledNames.has(server.name); + const forced = mcpForcedEnabled.has(server.name); + const sourceSaysDisabled = server.enabled === false && !forced; + const isDisabled = mcpDisabledNames.has(server.name) || disabledExtensions.has(id) || sourceSaysDisabled; const isShadowed = (server as { _shadowed?: boolean })._shadowed; const providerEnabled = isProviderEnabled(server._source.provider); diff --git a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts index 84c822a5f..54c670922 100644 --- a/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts +++ b/packages/coding-agent/test/extension-dashboard-mcp-parity.test.ts @@ -16,7 +16,7 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { initializeWithSettings } from "@oh-my-pi/pi-coding-agent/discovery"; +import { initializeWithSettings, reset as resetDiscoveryCache } from "@oh-my-pi/pi-coding-agent/discovery"; import { readMCPConfigFile, setMcpServerEnabled, setServerDisabled } from "@oh-my-pi/pi-coding-agent/mcp/config-writer"; import { loadAllExtensions } from "@oh-my-pi/pi-coding-agent/modes/components/extensions/state-manager"; import { __resetDirsFromEnvForTests, getMCPConfigPath, removeWithRetries, setAgentDir } from "@oh-my-pi/pi-utils"; @@ -218,6 +218,73 @@ describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => { expect(reenabled).toBeDefined(); expect(reenabled!.state).toBe("active"); }); + test("dashboard re-enable force-enables a tool-owned source (opencode.json) via enabledServers", async () => { + // OpenCode is a non-writable source: the dashboard must NOT mutate + // opencode.json, but the user-level enabledServers allowlist still has + // to flip the row active. Modeled after the codex review on PR #3829. + const opencodePath = path.join(projectDir, "opencode.json"); + await Bun.write( + opencodePath, + JSON.stringify({ + mcp: { + "opencode-server": { + type: "local", + command: ["echo", "opencode"], + enabled: false, + }, + }, + }), + ); + // beforeEach's Settings.init() already cached an absent opencode.json + // for this projectDir, so drop the capability fs cache before the first + // dashboard load picks the file up. + resetDiscoveryCache(); + + const before = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:opencode-server"); + expect(before).toBeDefined(); + expect(before!.source.provider).toBe("opencode"); + expect(before!.state).toBe("disabled"); + + // The dashboard withholds sourcePath for tool-owned sources, mirroring + // the #writableMcpSourcePath gate. + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "opencode-server", + enabled: true, + }); + + // opencode.json MUST stay untouched. + const opencodeRaw = JSON.parse(await Bun.file(opencodePath).text()) as { + mcp: { "opencode-server": { enabled: boolean } }; + }; + expect(opencodeRaw.mcp["opencode-server"].enabled).toBe(false); + + // The override lands in the user mcp.json's enabledServers list. + const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfig.enabledServers ?? []).toContain("opencode-server"); + + const after = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:opencode-server"); + expect(after).toBeDefined(); + expect(after!.state).toBe("active"); + + // Disabling again clears the override. + await setMcpServerEnabled({ + userPath: getMCPConfigPath("user", projectDir), + projectPath: getMCPConfigPath("project", projectDir), + name: "opencode-server", + enabled: false, + }); + + const userConfigAfter = await readMCPConfigFile(getMCPConfigPath("user", projectDir)); + expect(userConfigAfter.enabledServers ?? []).not.toContain("opencode-server"); + expect(userConfigAfter.disabledServers ?? []).toContain("opencode-server"); + + const offAgain = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:opencode-server"); + expect(offAgain).toBeDefined(); + expect(offAgain!.state).toBe("disabled"); + }); + test("dashboard toggles on a discovered (config-less) server use the denylist", async () => { // `phantom-server` is not in any config; only the denylist can suppress it. await setMcpServerEnabled({ From 3104d232aad37fc5f91b7d3714e3246817d28881 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 22:21:24 +0000 Subject: [PATCH 05/88] fix(irc): surfaced pending asides in inbox - Drained running-session IRC asides through the inbox tool before the model step consumes them. - Added a regression test for messages delivered while the recipient is already running. Fixes #3834 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/session/agent-session.ts | 43 +++++++++++++++++++ packages/coding-agent/src/tools/irc.ts | 12 ++++-- packages/coding-agent/test/tools/irc.test.ts | 23 ++++++++++ 4 files changed, 79 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..d4d841a28 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `irc` inbox drains missing messages that arrived while the recipient agent was already running. ([#3834](https://github.com/can1357/oh-my-pi/issues/3834)) + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 049d8ac03..d426af175 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -12839,6 +12839,49 @@ export class AgentSession { // IRC Delivery // ========================================================================= + /** + * Drains IRC incoming asides that have reached this running session but have + * not yet been folded into the next model step. + */ + drainPendingIrcInboxMessages(agentId: string, opts?: { peek?: boolean }): IrcMessage[] { + const messages: IrcMessage[] = []; + const remaining: CustomMessage[] = []; + for (const record of this.#pendingIrcAsides) { + if (record.customType !== "irc:incoming") { + remaining.push(record); + continue; + } + const details = record.details; + if (!details || typeof details !== "object") { + remaining.push(record); + continue; + } + const id = Reflect.get(details, "id"); + const from = Reflect.get(details, "from"); + const body = Reflect.get(details, "message"); + const replyTo = Reflect.get(details, "replyTo"); + if (typeof id !== "string" || typeof from !== "string" || typeof body !== "string") { + remaining.push(record); + continue; + } + messages.push({ + id, + from, + to: agentId, + body, + ts: record.timestamp, + ...(typeof replyTo === "string" ? { replyTo } : {}), + }); + if (opts?.peek) { + remaining.push(record); + } + } + if (!opts?.peek) { + this.#pendingIrcAsides = remaining; + } + return messages; + } + /** * Deliver an IRC message into this session (recipient side; called by the * IrcBus). Emits the `irc_message` session event for UI cards and injects diff --git a/packages/coding-agent/src/tools/irc.ts b/packages/coding-agent/src/tools/irc.ts index f37c91e58..6993a768f 100644 --- a/packages/coding-agent/src/tools/irc.ts +++ b/packages/coding-agent/src/tools/irc.ts @@ -172,7 +172,7 @@ export class IrcTool implements AgentTool { case "wait": return this.#executeWait(senderId, params, signal); case "inbox": - return this.#executeInbox(senderId, params); + return this.#executeInbox(registry, senderId, params); default: return errorResult("Unknown irc op.", { op: params.op }); } @@ -371,8 +371,14 @@ export class IrcTool implements AgentTool { }; } - #executeInbox(senderId: string, params: IrcParams): AgentToolResult { - const messages = IrcBus.global().inbox(senderId, { peek: params.peek }); + #executeInbox(registry: AgentRegistry, senderId: string, params: IrcParams): AgentToolResult { + const busMessages = IrcBus.global().inbox(senderId, { peek: params.peek }); + const session = registry.get(senderId)?.session; + const pendingMessages = + typeof session?.drainPendingIrcInboxMessages === "function" + ? session.drainPendingIrcInboxMessages(senderId, { peek: params.peek }) + : []; + const messages = [...busMessages, ...pendingMessages].sort((a, b) => a.ts - b.ts); if (messages.length === 0) { return { content: [{ type: "text", text: "Inbox empty." }], diff --git a/packages/coding-agent/test/tools/irc.test.ts b/packages/coding-agent/test/tools/irc.test.ts index 47c35239e..ed7d8885d 100644 --- a/packages/coding-agent/test/tools/irc.test.ts +++ b/packages/coding-agent/test/tools/irc.test.ts @@ -580,6 +580,29 @@ describe("IRC", () => { expect(text).toContain("No message"); }); + it("op=inbox drains IRC asides that arrived while the caller was running", async () => { + const { session } = createRealSession(); + sessions.push(session); + Object.defineProperty(session, "isStreaming", { value: true, configurable: true }); + registry.register({ id: "0-Running", displayName: "task", kind: "sub", session }); + + const delivery = await session.deliverIrcMessage({ + id: "msg-running", + from: "0-Main", + to: "0-Running", + body: "parallel note", + ts: Date.now(), + }); + expect(delivery).toBe("injected"); + + const tool = new IrcTool(makeToolSession(registry, "0-Running")); + const result = await tool.execute("call-1", { op: "inbox" }); + + expect(result.details?.inbox?.map(msg => msg.body)).toEqual(["parallel note"]); + const text = result.content[0]?.type === "text" ? result.content[0].text : ""; + expect(text).toContain("parallel note"); + }); + it("op=inbox drains the caller's mailbox", async () => { const main = makeFakeSession(); registry.register({ id: "0-Main", displayName: "main", kind: "main", session: main.session }); From 66eba507f543d74ef28949c6b6ef2943ac634492 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 22:29:10 +0000 Subject: [PATCH 06/88] fix(irc): consumed peeked pending asides to avoid double inject - Always remove surfaced irc:incoming records from the pending-aside queue; the inbox tool result already injects the body, so leaving them queued would auto-inject a duplicate at the next step. - Updated the inbox tool to drain pending asides regardless of peek. - Added a regression test asserting a peeked pending aside does not auto-inject. --- .../coding-agent/src/session/agent-session.ts | 19 +++++++------- packages/coding-agent/src/tools/irc.ts | 2 +- packages/coding-agent/test/tools/irc.test.ts | 26 +++++++++++++++++++ 3 files changed, 37 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index d426af175..a3a3de721 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -12840,10 +12840,16 @@ export class AgentSession { // ========================================================================= /** - * Drains IRC incoming asides that have reached this running session but have - * not yet been folded into the next model step. + * Surfaces (and consumes) IRC incoming asides that have reached this running + * session but have not yet been folded into the next model step. + * + * The inbox tool injects the formatted body into the tool result, so the + * model sees it once via the result. Leaving the record in + * {@link #pendingIrcAsides} would let the aside provider deliver it a second + * time at the next step boundary — including on `peek`, which is why peek + * also drains here. */ - drainPendingIrcInboxMessages(agentId: string, opts?: { peek?: boolean }): IrcMessage[] { + drainPendingIrcInboxMessages(agentId: string): IrcMessage[] { const messages: IrcMessage[] = []; const remaining: CustomMessage[] = []; for (const record of this.#pendingIrcAsides) { @@ -12872,13 +12878,8 @@ export class AgentSession { ts: record.timestamp, ...(typeof replyTo === "string" ? { replyTo } : {}), }); - if (opts?.peek) { - remaining.push(record); - } - } - if (!opts?.peek) { - this.#pendingIrcAsides = remaining; } + this.#pendingIrcAsides = remaining; return messages; } diff --git a/packages/coding-agent/src/tools/irc.ts b/packages/coding-agent/src/tools/irc.ts index 6993a768f..1f483f048 100644 --- a/packages/coding-agent/src/tools/irc.ts +++ b/packages/coding-agent/src/tools/irc.ts @@ -376,7 +376,7 @@ export class IrcTool implements AgentTool { const session = registry.get(senderId)?.session; const pendingMessages = typeof session?.drainPendingIrcInboxMessages === "function" - ? session.drainPendingIrcInboxMessages(senderId, { peek: params.peek }) + ? session.drainPendingIrcInboxMessages(senderId) : []; const messages = [...busMessages, ...pendingMessages].sort((a, b) => a.ts - b.ts); if (messages.length === 0) { diff --git a/packages/coding-agent/test/tools/irc.test.ts b/packages/coding-agent/test/tools/irc.test.ts index ed7d8885d..48788f4aa 100644 --- a/packages/coding-agent/test/tools/irc.test.ts +++ b/packages/coding-agent/test/tools/irc.test.ts @@ -603,6 +603,32 @@ describe("IRC", () => { expect(text).toContain("parallel note"); }); + it("op=inbox peek surfaces a pending IRC aside and prevents it auto-injecting", async () => { + const { session } = createRealSession(); + sessions.push(session); + Object.defineProperty(session, "isStreaming", { value: true, configurable: true }); + registry.register({ id: "0-Running", displayName: "task", kind: "sub", session }); + + await session.deliverIrcMessage({ + id: "msg-peek", + from: "0-Main", + to: "0-Running", + body: "peeked note", + ts: Date.now(), + }); + + const tool = new IrcTool(makeToolSession(registry, "0-Running")); + const peeked = await tool.execute("call-1", { op: "inbox", peek: true }); + expect(peeked.details?.inbox?.map(msg => msg.body)).toEqual(["peeked note"]); + + // The peek surfaced the body via the tool result, so the aside-channel + // copy must NOT also be auto-injected at the next step: a second drain + // returns nothing (the pending aside was consumed out of the + // auto-inject queue when peek surfaced it). + const second = await tool.execute("call-2", { op: "inbox" }); + expect(second.details?.inbox).toEqual([]); + }); + it("op=inbox drains the caller's mailbox", async () => { const main = makeFakeSession(); registry.register({ id: "0-Main", displayName: "main", kind: "main", session: main.session }); From 2933f8f32ce80e9389aebff06c6585bc7bd913e1 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 22:38:20 +0000 Subject: [PATCH 07/88] fix(prompting): bounded gpu detection Cached empty GPU probe results and routed GPU detection through the system prompt prep deadline so slow WMI/lspci probes cannot block startup indefinitely.\n\nFixes #3835 --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/system-prompt.test.ts | 100 ++++++++++++++++++ packages/coding-agent/src/system-prompt.ts | 33 +++--- 3 files changed, 121 insertions(+), 16 deletions(-) create mode 100644 packages/coding-agent/src/system-prompt.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..09775f651 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed system-prompt GPU detection blocking startup and rerunning failed probes by applying the prep deadline and caching empty results. ([#3835](https://github.com/can1357/oh-my-pi/issues/3835)) + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts new file mode 100644 index 000000000..8196e207a --- /dev/null +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -0,0 +1,100 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; + +interface ProbeRunResult { + elapsedMs: number; + cached: unknown; + count: number; +} + +async function runProbeScenario(options: { runs: number; sleepSeconds?: number }): Promise { + const tempRoot = await fs.mkdtemp(path.join(os.tmpdir(), "omp-gpu-probe-")); + try { + const binDir = path.join(tempRoot, "bin"); + const cacheRoot = path.join(tempRoot, "cache"); + const probeCountPath = path.join(tempRoot, "probe-count"); + await fs.mkdir(binDir, { recursive: true }); + await fs.mkdir(path.join(cacheRoot, "omp"), { recursive: true }); + const lspciPath = path.join(binDir, "lspci"); + await Bun.write( + lspciPath, + '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', + ); + await fs.chmod(lspciPath, 0o755); + + const scenarioPath = path.join(tempRoot, "scenario.ts"); + await Bun.write( + scenarioPath, + `import { getGpuCachePath, refreshDirsFromEnv } from ${JSON.stringify(path.resolve(import.meta.dir, "../../utils/src/index.ts"))}; +import { buildSystemPrompt } from ${JSON.stringify(path.join(import.meta.dir, "system-prompt.ts"))}; + +refreshDirsFromEnv(); +const buildOptions = { + contextFiles: [], + skills: [], + toolNames: [], + workspaceTree: { + rootPath: process.cwd(), + rendered: "", + truncated: false, + totalLines: 0, + agentsMdFiles: [], + }, + activeRepoContext: null, +}; +const startedAt = performance.now(); +for (let index = 0; index < Number(process.env.OMP_GPU_PROBE_RUNS ?? "1"); index += 1) { + await buildSystemPrompt(buildOptions); +} +const cacheFile = Bun.file(getGpuCachePath()); +const cached = await cacheFile.exists() ? await cacheFile.json() : null; +const countFile = Bun.file(process.env.OMP_GPU_PROBE_COUNT ?? ""); +const count = await countFile.exists() ? (await countFile.text()).length : 0; +console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt), cached, count })); +`, + ); + + const env: Record = { + ...process.env, + PATH: `${binDir}:${process.env.PATH ?? ""}`, + XDG_CACHE_HOME: cacheRoot, + OMP_GPU_PROBE_COUNT: probeCountPath, + OMP_GPU_PROBE_RUNS: String(options.runs), + }; + if (options.sleepSeconds === undefined) { + delete env.OMP_GPU_PROBE_SLEEP; + } else { + env.OMP_GPU_PROBE_SLEEP = String(options.sleepSeconds); + } + + const child = Bun.spawn([process.execPath, scenarioPath], { stdout: "pipe", stderr: "pipe", env }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(child.stdout).text(), + new Response(child.stderr).text(), + child.exited, + ]); + if (exitCode !== 0) { + throw new Error(`GPU probe scenario failed with exit ${exitCode}: ${stderr}`); + } + return JSON.parse(stdout.trim()); + } finally { + await fs.rm(tempRoot, { recursive: true, force: true }); + } +} + +describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { + it("caches empty GPU probe results", async () => { + const result = await runProbeScenario({ runs: 2 }); + + expect(result.cached).toEqual({ gpu: null }); + expect(result.count).toBe(1); + }, 15_000); + + it("uses the system prompt prep deadline for slow GPU probes", async () => { + const result = await runProbeScenario({ runs: 1, sleepSeconds: 7 }); + + expect(result.elapsedMs).toBeLessThan(6500); + }, 15_000); +}); diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 35048ba5b..fc616a519 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -176,20 +176,20 @@ function getTerminalName(): string | undefined { return term ?? undefined; } -/** Cached system info structure */ +/** Cached GPU probe result. */ interface GpuCache { - gpu: string; -} - -function getSystemInfoCachePath(): string { - return getGpuCachePath(); + gpu: string | null; } async function loadGpuCache(): Promise { try { - const cachePath = getSystemInfoCachePath(); + const cachePath = getGpuCachePath(); const content = await Bun.file(cachePath).json(); - return content as GpuCache; + if (content && typeof content === "object" && "gpu" in content) { + const gpu = content.gpu; + return { gpu: typeof gpu === "string" ? gpu : null }; + } + return null; } catch { return null; } @@ -197,7 +197,7 @@ async function loadGpuCache(): Promise { async function saveGpuCache(info: GpuCache): Promise { try { - const cachePath = getSystemInfoCachePath(); + const cachePath = getGpuCachePath(); await Bun.write(cachePath, JSON.stringify(info, null, "\t")); } catch { // Silently ignore cache write failures @@ -206,15 +206,12 @@ async function saveGpuCache(info: GpuCache): Promise { async function getCachedGpu(): Promise { const cached = await logger.time("getCachedGpu:loadGpuCache", loadGpuCache); - if (cached) return cached.gpu; + if (cached) return cached.gpu ?? undefined; const gpu = await logger.time("getCachedGpu:getGpuModel", getGpuModel); - if (gpu) { - await logger.time("getCachedGpu:saveGpuCache", saveGpuCache, { gpu }); - } + await logger.time("getCachedGpu:saveGpuCache", saveGpuCache, { gpu }); return gpu ?? undefined; } -async function getEnvironmentInfo(): Promise> { - const gpu = await getCachedGpu(); +function getEnvironmentInfo(gpu: string | undefined): Array<{ label: string; value: string }> { let cpuModel: string | undefined; try { cpuModel = os.cpus()[0]?.model; @@ -500,6 +497,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): agentsMdFiles: [], } satisfies WorkspaceTree, activeRepoContext: null as ActiveRepoContext | null, + gpu: undefined as string | undefined, }; const deadline = Bun.sleep(SYSTEM_PROMPT_PREP_TIMEOUT_MS).then(() => "__timeout__" as const); @@ -566,6 +564,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): providedActiveRepoContext !== undefined ? Promise.resolve(providedActiveRepoContext) : logger.time("resolveActiveRepoContext", () => resolveActiveRepoContext(resolvedCwd)); + const gpuPromise = logger.time("getCachedGpu", getCachedGpu); const [ resolvedCustomPrompt, @@ -575,6 +574,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): skills, workspaceTree, activeRepoContext, + gpu, ] = await Promise.all([ withDeadline( "customPrompt", @@ -597,6 +597,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): withDeadline("loadSkills", skillsPromise, prepDefaults.skills), withDeadline("buildWorkspaceTree", workspaceTreePromise, prepDefaults.workspaceTree), withDeadline("resolveActiveRepoContext", activeRepoContextPromise, prepDefaults.activeRepoContext), + withDeadline("getCachedGpu", gpuPromise, prepDefaults.gpu), ]); const agentsMdFiles = Array.from(new Set(workspaceTree.agentsMdFiles)).sort().slice(0, AGENTS_MD_LIMIT); @@ -675,7 +676,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): ]; const injectedAlwaysApplyRules = dedupeAlwaysApplyRules(alwaysApplyRules, promptSources); - const environment = await logger.time("getEnvironmentInfo", getEnvironmentInfo); + const environment = getEnvironmentInfo(gpu); const data = { systemPromptCustomization: effectiveSystemPromptCustomization, customPrompt: resolvedCustomPrompt, From 089708b219556d6600d1f1415e99e4fc635f38ec Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 22:44:42 +0000 Subject: [PATCH 08/88] test(prompting): isolated gpu probe from inherited profile env Stripped PI_CODING_AGENT_DIR, OMP_PROFILE, PI_PROFILE and PI_CONFIG_DIR from the GPU probe child environment so the temporary XDG_CACHE_HOME wins instead of resolving the developer/CI profile's real gpu_cache.json. --- packages/coding-agent/src/system-prompt.test.ts | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index 8196e207a..9ab8f7092 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -63,6 +63,11 @@ console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt OMP_GPU_PROBE_COUNT: probeCountPath, OMP_GPU_PROBE_RUNS: String(options.runs), }; + // Strip inherited dirs-resolver overrides so XDG_CACHE_HOME above wins and + // the test cannot touch the developer/CI profile's real gpu_cache.json. + for (const key of ["PI_CODING_AGENT_DIR", "OMP_PROFILE", "PI_PROFILE", "PI_CONFIG_DIR"]) { + delete env[key]; + } if (options.sleepSeconds === undefined) { delete env.OMP_GPU_PROBE_SLEEP; } else { From 7e7b769e455989fc518e274baaf23cbbe8fa8018 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 22:54:23 +0000 Subject: [PATCH 09/88] fix(prompting): killed wedged gpu probe at the prep deadline Replaced the Bun Shell call with Bun.spawn + timeout, so a hanging lspci/wmic now receives SIGTERM at the prep deadline instead of keeping the Bun process alive after withDeadline returned. Tightened the regression test to also assert the spawned scenario process exits within the deadline. --- .../coding-agent/src/system-prompt.test.ts | 12 ++++++-- packages/coding-agent/src/system-prompt.ts | 29 +++++++++++++------ 2 files changed, 29 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index 9ab8f7092..95560f240 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -5,6 +5,7 @@ import * as path from "node:path"; interface ProbeRunResult { elapsedMs: number; + childElapsedMs: number; cached: unknown; count: number; } @@ -20,7 +21,7 @@ async function runProbeScenario(options: { runs: number; sleepSeconds?: number } const lspciPath = path.join(binDir, "lspci"); await Bun.write( lspciPath, - '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', + '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', ); await fs.chmod(lspciPath, 0o755); @@ -74,16 +75,18 @@ console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt env.OMP_GPU_PROBE_SLEEP = String(options.sleepSeconds); } + const childStartedAt = performance.now(); const child = Bun.spawn([process.execPath, scenarioPath], { stdout: "pipe", stderr: "pipe", env }); const [stdout, stderr, exitCode] = await Promise.all([ new Response(child.stdout).text(), new Response(child.stderr).text(), child.exited, ]); + const childElapsedMs = Math.round(performance.now() - childStartedAt); if (exitCode !== 0) { throw new Error(`GPU probe scenario failed with exit ${exitCode}: ${stderr}`); } - return JSON.parse(stdout.trim()); + return { ...JSON.parse(stdout.trim()), childElapsedMs }; } finally { await fs.rm(tempRoot, { recursive: true, force: true }); } @@ -97,9 +100,12 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { expect(result.count).toBe(1); }, 15_000); - it("uses the system prompt prep deadline for slow GPU probes", async () => { + it("kills the GPU probe at the prep deadline", async () => { const result = await runProbeScenario({ runs: 1, sleepSeconds: 7 }); expect(result.elapsedMs).toBeLessThan(6500); + // Codex#3838: the child process MUST exit shortly after the deadline, + // not linger until the underlying probe (sleep 7) finishes on its own. + expect(result.childElapsedMs).toBeLessThan(6500); }, 15_000); }); diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index fc616a519..2bce76add 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -7,7 +7,6 @@ import type { AgentTool } from "@oh-my-pi/pi-agent-core"; import type { ToolExample, TSchema } from "@oh-my-pi/pi-ai"; import { renderToolInventory } from "@oh-my-pi/pi-ai/dialect"; import { $env, getGpuCachePath, getProjectDir, hasFsCode, isEnoent, logger, prompt } from "@oh-my-pi/pi-utils"; -import { $ } from "bun"; import { contextFileCapability } from "./capability/context-file"; import { systemPromptCapability } from "./capability/system-prompt"; import { findConfigFile } from "./config"; @@ -112,21 +111,33 @@ function parseWmicTable(output: string, header: string): string | null { } const SYSTEM_PROMPT_PREP_TIMEOUT_MS = 5000; +/** Killed by the OS at this deadline, so a wedged probe cannot outlive the prep race. */ +const GPU_PROBE_TIMEOUT_MS = SYSTEM_PROMPT_PREP_TIMEOUT_MS; + +async function runGpuProbe(cmd: string[]): Promise { + try { + const proc = Bun.spawn({ + cmd, + stdout: "pipe", + stderr: "ignore", + stdin: "ignore", + timeout: GPU_PROBE_TIMEOUT_MS, + }); + const [stdout, exitCode] = await Promise.all([new Response(proc.stdout).text(), proc.exited]); + return exitCode === 0 ? stdout : null; + } catch { + return null; + } +} async function getGpuModel(): Promise { switch (process.platform) { case "win32": { - const output = await $`wmic path win32_VideoController get name` - .quiet() - .text() - .catch(() => null); + const output = await runGpuProbe(["wmic", "path", "win32_VideoController", "get", "name"]); return output ? parseWmicTable(output, "Name") : null; } case "linux": { - const output = await $`lspci` - .quiet() - .text() - .catch(() => null); + const output = await runGpuProbe(["lspci"]); if (!output) return null; const gpus: Array<{ name: string; priority: number }> = []; for (const line of output.split("\n")) { From f5be8953c420eb20bf840a8b0c50a7cd7578fe6d Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 22:56:40 +0000 Subject: [PATCH 10/88] fix(cli): surfaced tiny model download errors Preserved worker-side download errors through TinyTitleClient and included them in tiny-models text and JSON failures. Fixes #3839 --- packages/coding-agent/CHANGELOG.md | 4 +++ .../coding-agent/src/cli/tiny-models-cli.ts | 22 ++++++++++--- .../coding-agent/src/tiny/title-client.ts | 32 +++++++++++-------- .../test/issue-1940-repro.test.ts | 22 ++++++++++++- .../coding-agent/test/tiny-models-cli.test.ts | 29 +++++++++++++++-- 5 files changed, 89 insertions(+), 20 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..249e07b6a 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `omp tiny-models download` JSON/text failures to include the worker-side download error instead of collapsing every worker failure to `ok:false`. ([#3839](https://github.com/can1357/oh-my-pi/issues/3839)) + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/cli/tiny-models-cli.ts b/packages/coding-agent/src/cli/tiny-models-cli.ts index 02a3be475..b2efc23b0 100644 --- a/packages/coding-agent/src/cli/tiny-models-cli.ts +++ b/packages/coding-agent/src/cli/tiny-models-cli.ts @@ -28,12 +28,21 @@ interface ProgressReporter { interface DownloadResult { model: TinyLocalModelKey; ok: boolean; + error?: string; } function writeLine(text = ""): void { process.stdout.write(`${text}\n`); } +function downloadErrorSummary(error: string | undefined): string | undefined { + return error + ?.split(/\r?\n/) + .map(line => line.trim()) + .find(line => line.length > 0) + ?.replace(/^Error:\s*/, ""); +} + export function resolveModels(model: string | undefined): TinyLocalModelKey[] { if (!model) return [DEFAULT_TINY_TITLE_LOCAL_MODEL_KEY]; // `all` is a prefetch convenience: skip models that fail before load (unsupported @@ -101,10 +110,15 @@ async function downloadOne(modelKey: TinyLocalModelKey, json: boolean | undefine const label = getTinyLocalModelSpec(modelKey)?.label ?? modelKey; if (!json && !process.stdout.isTTY) writeLine(`Downloading ${label} (${modelKey})...`); const progress = makeProgressReporter(modelKey, json); - const ok = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress }); - progress.finish(ok); - if (!json && !process.stdout.isTTY) writeLine(ok ? `Downloaded ${label}.` : `Failed to download ${label}.`); - return { model: modelKey, ok }; + const result = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress }); + progress.finish(result.ok); + const error = downloadErrorSummary(result.error); + if (!json && !process.stdout.isTTY) { + writeLine(result.ok ? `Downloaded ${label}.` : `Failed to download ${label}${error ? `: ${error}` : ""}.`); + } else if (!json && !result.ok && error) { + writeLine(`${label} failed: ${error}`); + } + return result.error ? { model: modelKey, ok: result.ok, error: result.error } : { model: modelKey, ok: result.ok }; } export async function runTinyModelsCommand(command: TinyModelsCommandArgs): Promise { diff --git a/packages/coding-agent/src/tiny/title-client.ts b/packages/coding-agent/src/tiny/title-client.ts index 5ea9c9bab..b2d31b6d3 100644 --- a/packages/coding-agent/src/tiny/title-client.ts +++ b/packages/coding-agent/src/tiny/title-client.ts @@ -29,7 +29,12 @@ import type { TinyTitleProgressEvent, TinyTitleWorkerInbound, TinyTitleWorkerOut type PendingRequest = | { kind: "generate"; modelKey: TinyTitleLocalModelKey; resolve: (title: string | null) => void } | { kind: "complete"; modelKey: TinyMemoryLocalModelKey; resolve: (text: string | null) => void } - | { kind: "download"; modelKey: TinyLocalModelKey; resolve: (ok: boolean) => void }; + | { kind: "download"; modelKey: TinyLocalModelKey; resolve: (result: TinyTitleDownloadResult) => void }; + +export interface TinyTitleDownloadResult { + ok: boolean; + error?: string; +} export interface TinyTitleDownloadOptions { signal?: AbortSignal; @@ -269,21 +274,21 @@ export class TinyTitleClient { } } - async downloadModel(modelKey: string, options: TinyTitleDownloadOptions = {}): Promise { - if (!isTinyLocalModelKey(modelKey)) return false; - if (options.signal?.aborted) return false; + async downloadModel(modelKey: string, options: TinyTitleDownloadOptions = {}): Promise { + if (!isTinyLocalModelKey(modelKey)) return { ok: false }; + if (options.signal?.aborted) return { ok: false }; const unsubscribe = options.onProgress ? this.onProgress(options.onProgress) : undefined; try { const worker = this.#ensureWorker(); const id = String(++this.#nextRequestId); - const { promise, resolve } = Promise.withResolvers(); + const { promise, resolve } = Promise.withResolvers(); this.#addPending(id, { kind: "download", modelKey, resolve }); const abort = (): void => { const pending = this.#pending.get(id); if (pending?.kind !== "download") return; this.#deletePending(id); - pending.resolve(false); + pending.resolve({ ok: false }); }; options.signal?.addEventListener("abort", abort, { once: true }); try { @@ -294,11 +299,12 @@ export class TinyTitleClient { this.#deletePending(id); } } catch (error) { + const message = error instanceof Error ? error.message : String(error); logger.debug("tiny-title: local model download failed", { modelKey, - error: error instanceof Error ? error.message : String(error), + error: message, }); - return false; + return { ok: false, error: message }; } finally { unsubscribe?.(); } @@ -314,7 +320,7 @@ export class TinyTitleClient { for (const pending of this.#pending.values()) { this.#emitProgress({ modelKey: pending.modelKey, status: "error" }); if (pending.kind === "generate" || pending.kind === "complete") pending.resolve(null); - else pending.resolve(false); + else pending.resolve({ ok: false }); } this.#pending.clear(); this.#refed = false; @@ -379,7 +385,7 @@ export class TinyTitleClient { return; } if (message.type === "downloaded") { - if (pending.kind === "download") pending.resolve(true); + if (pending.kind === "download") pending.resolve({ ok: true }); return; } if (message.type === "completion") { @@ -389,8 +395,8 @@ export class TinyTitleClient { logger.debug("tiny-title: worker returned error", { error: message.error }); this.#markFailedModel(pending); this.#emitProgress({ modelKey: pending.modelKey, status: "error" }); - if (pending.kind === "generate" || pending.kind === "complete") pending.resolve(null); - else pending.resolve(false); + if (pending.kind === "download") pending.resolve({ ok: false, error: message.error }); + else pending.resolve(null); void this.terminate(); } @@ -407,7 +413,7 @@ export class TinyTitleClient { for (const pending of this.#pending.values()) { this.#emitProgress({ modelKey: pending.modelKey, status: "error" }); if (pending.kind === "generate" || pending.kind === "complete") pending.resolve(null); - else pending.resolve(false); + else pending.resolve({ ok: false, error: error.message }); } this.#pending.clear(); void this.terminate(); diff --git a/packages/coding-agent/test/issue-1940-repro.test.ts b/packages/coding-agent/test/issue-1940-repro.test.ts index c89aed721..4568d1b48 100644 --- a/packages/coding-agent/test/issue-1940-repro.test.ts +++ b/packages/coding-agent/test/issue-1940-repro.test.ts @@ -171,10 +171,30 @@ describe("issue #3291 — tiny-model downloads keep the worker referenced", () = worker.emit({ type: "downloaded", id: downloadRequestId }); - expect(await download).toBe(true); + expect(await download).toEqual({ ok: true }); expect(worker.unrefCalls).toBe(1); } finally { await client.terminate(); } }); + + it("returns the worker error for failed download requests", async () => { + let downloadRequestId = ""; + const worker = new FakeTinyWorker(message => { + if (message.type === "download") downloadRequestId = message.id; + }); + const client = new TinyTitleClient(() => worker); + + try { + const download = client.downloadModel("lfm2-700m"); + + expect(downloadRequestId).not.toBe(""); + worker.emit({ type: "error", id: downloadRequestId, error: "Error: runtime install failed" }); + + expect(await download).toEqual({ ok: false, error: "Error: runtime install failed" }); + expect(worker.terminated).toBe(true); + } finally { + await client.terminate(); + } + }); }); diff --git a/packages/coding-agent/test/tiny-models-cli.test.ts b/packages/coding-agent/test/tiny-models-cli.test.ts index c81e49006..363d1eeba 100644 --- a/packages/coding-agent/test/tiny-models-cli.test.ts +++ b/packages/coding-agent/test/tiny-models-cli.test.ts @@ -1,6 +1,11 @@ -import { describe, expect, it } from "bun:test"; -import { resolveModels } from "@oh-my-pi/pi-coding-agent/cli/tiny-models-cli"; +import { afterEach, describe, expect, it, spyOn, vi } from "bun:test"; +import { resolveModels, runTinyModelsCommand } from "@oh-my-pi/pi-coding-agent/cli/tiny-models-cli"; import { TINY_LOCAL_MODELS } from "@oh-my-pi/pi-coding-agent/tiny/models"; +import { tinyTitleClient } from "@oh-my-pi/pi-coding-agent/tiny/title-client"; + +afterEach(() => { + vi.restoreAllMocks(); +}); describe("tiny-models download model resolution", () => { it("excludes load-blocked models from `all` so the bulk prefetch stays green", () => { @@ -25,4 +30,24 @@ describe("tiny-models download model resolution", () => { if (!blocked) return; expect(resolveModels(blocked.key)).toEqual([blocked.key]); }); + + it("includes worker error details in JSON failures", async () => { + const output: string[] = []; + spyOn(process.stdout, "write").mockImplementation((chunk: string | Uint8Array) => { + output.push(typeof chunk === "string" ? chunk : new TextDecoder().decode(chunk)); + return true; + }); + spyOn(tinyTitleClient, "downloadModel").mockResolvedValue({ + ok: false, + error: "Error: runtime install failed\n at worker", + }); + + await expect( + runTinyModelsCommand({ action: "download", model: "lfm2-700m", flags: { json: true } }), + ).rejects.toThrow("One or more tiny title models failed to download"); + + expect(JSON.parse(output.join(""))).toEqual({ + results: [{ model: "lfm2-700m", ok: false, error: "Error: runtime install failed\n at worker" }], + }); + }); }); From fd070bf3aa6fd1254526e2321995a57a1681c844 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 23:33:46 +0000 Subject: [PATCH 11/88] fix(prompting): killed gpu probe with sigkill on timeout Bun.spawn defaults killSignal to SIGTERM, which a wedged lspci/wmic (or its PATH wrapper) can ignore, leaving the probe alive past the prep deadline and blocking the null-cache write. Force SIGKILL so proc.exited always resolves at the deadline and the next startup hits the cache. --- packages/coding-agent/src/system-prompt.ts | 3 +++ 1 file changed, 3 insertions(+) diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 2bce76add..c5b959a0c 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -122,6 +122,9 @@ async function runGpuProbe(cmd: string[]): Promise { stderr: "ignore", stdin: "ignore", timeout: GPU_PROBE_TIMEOUT_MS, + // SIGKILL so a probe ignoring SIGTERM (PATH wrapper, wedged WMI) still + // dies at the deadline and lets getCachedGpu reach the null-cache write. + killSignal: "SIGKILL", }); const [stdout, exitCode] = await Promise.all([new Response(proc.stdout).text(), proc.exited]); return exitCode === 0 ? stdout : null; From 4b98211c64681e0415f9880595aa70b03a95bfc3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 00:09:34 +0000 Subject: [PATCH 12/88] fix(coding-agent): fixed dirty isolated branch merges Applied isolated branch patches with three-way fallback when unrelated parent dirt appears in patch context. Surfaced branch preparation failures instead of reporting no changes. Fixes #3841 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/task/isolation-runner.ts | 8 +++ packages/coding-agent/src/task/worktree.ts | 52 +++++++++++++------ packages/coding-agent/src/utils/git.ts | 2 + .../test/task/isolation-runner.test.ts | 19 +++++++ .../coding-agent/test/task/worktree.test.ts | 40 ++++++++++++++ 6 files changed, 108 insertions(+), 17 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..d91d1c260 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed isolated branch merges rejecting task edits when the parent checkout had unrelated dirty changes in nearby patch context. ([#3841](https://github.com/can1357/oh-my-pi/issues/3841)) + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/task/isolation-runner.ts b/packages/coding-agent/src/task/isolation-runner.ts index 455bba052..c62a3d5fa 100644 --- a/packages/coding-agent/src/task/isolation-runner.ts +++ b/packages/coding-agent/src/task/isolation-runner.ts @@ -222,6 +222,14 @@ export async function mergeIsolatedChanges(opts: IsolationMergeOptions): Promise const { result, repoRoot, mergeMode } = opts; try { if (mergeMode === "branch") { + if (!result.branchName && result.exitCode === 0 && !result.aborted && result.error) { + return { + summary: `\n\nBranch merge failed before a task branch could be created: ${result.error}\nTask outputs are preserved but changes were not applied.`, + changesApplied: false, + hadAnyChanges: false, + mergedBranchForNestedPatches: false, + }; + } const canApplyNestedOnly = !result.branchName && result.exitCode === 0 && !result.aborted && (result.nestedPatches?.length ?? 0) > 0; if (!result.branchName || result.exitCode !== 0 || result.aborted) { diff --git a/packages/coding-agent/src/task/worktree.ts b/packages/coding-agent/src/task/worktree.ts index 8dee8163d..31ca1f197 100644 --- a/packages/coding-agent/src/task/worktree.ts +++ b/packages/coding-agent/src/task/worktree.ts @@ -148,21 +148,33 @@ export async function captureBaseline(repoRoot: string): Promise { +async function captureRepoDeltaPatch(repoDir: string, rb: RepoBaseline, objectRepoDir = repoDir): Promise { const currentHead = (await git.head.sha(repoDir)) ?? ""; const currentStaged = await git.diff(repoDir, { binary: true, cached: true }); const currentUnstaged = await git.diff(repoDir, { binary: true }); const currentUntracked = await git.ls.untracked(repoDir); const currentUntrackedPatch = await captureUntrackedPatch(repoDir, currentUntracked); + const committedPatch = + currentHead && currentHead !== rb.headCommit + ? await git.diff.tree(repoDir, rb.headCommit, currentHead, { + allowFailure: true, + binary: true, + }) + : ""; - const baselineTree = await writeSyntheticTree(repoDir, rb.headCommit, [rb.staged, rb.unstaged, rb.untrackedPatch]); - const currentTree = await writeSyntheticTree(repoDir, currentHead, [ + const baselineTree = await writeSyntheticTree(objectRepoDir, rb.headCommit, [ + rb.staged, + rb.unstaged, + rb.untrackedPatch, + ]); + const currentTree = await writeSyntheticTree(objectRepoDir, rb.headCommit, [ + committedPatch, currentStaged, currentUnstaged, currentUntrackedPatch, ]); - return git.diff.tree(repoDir, baselineTree, currentTree, { + return git.diff.tree(objectRepoDir, baselineTree, currentTree, { allowFailure: true, binary: true, }); @@ -212,7 +224,7 @@ export interface DeltaPatchResult { } export async function captureDeltaPatch(isolationDir: string, baseline: WorktreeBaseline): Promise { - const rootPatch = await captureRepoDeltaPatch(isolationDir, baseline.root); + const rootPatch = await captureRepoDeltaPatch(isolationDir, baseline.root, baseline.root.repoRoot); const nestedPatches: NestedRepoPatch[] = []; for (const { relativePath, baseline: nb } of baseline.nested) { @@ -222,7 +234,7 @@ export async function captureDeltaPatch(isolationDir: string, baseline: Worktree } catch { continue; } - const patch = await captureRepoDeltaPatch(nestedDir, nb); + const patch = await captureRepoDeltaPatch(nestedDir, nb, nb.repoRoot); if (patch.trim()) nestedPatches.push({ relativePath, patch }); } @@ -490,18 +502,24 @@ export async function commitToBranch( try { await git.patch.applyText(tmpDir, rootPatch); } catch (err) { - if (err instanceof git.GitCommandError) { - const stderr = err.result.stderr.slice(0, 2000); - logger.error("commitToBranch: git apply failed", { - taskId, - exitCode: err.result.exitCode, - stderr, - patchSize: rootPatch.length, - patchHead: rootPatch.slice(0, 500), - }); - throw new Error(`git apply failed for task ${taskId}: ${stderr}`); + if (!(err instanceof git.GitCommandError)) throw err; + try { + await git.patch.applyText(tmpDir, rootPatch, { threeWay: true }); + } catch (threeWayErr) { + if (threeWayErr instanceof git.GitCommandError) { + const stderr = threeWayErr.result.stderr.slice(0, 2000); + logger.error("commitToBranch: git apply --3way failed", { + taskId, + exitCode: threeWayErr.result.exitCode, + stderr, + initialStderr: err.result.stderr.slice(0, 2000), + patchSize: rootPatch.length, + patchHead: rootPatch.slice(0, 500), + }); + throw new Error(`git apply --3way failed for task ${taskId}: ${stderr}`); + } + throw threeWayErr; } - throw err; } await git.stage.files(tmpDir); const msg = (commitMessage && (await commitMessage(rootPatch))) || fallbackMessage; diff --git a/packages/coding-agent/src/utils/git.ts b/packages/coding-agent/src/utils/git.ts index 86afc8bc0..dc5115ab7 100644 --- a/packages/coding-agent/src/utils/git.ts +++ b/packages/coding-agent/src/utils/git.ts @@ -91,6 +91,7 @@ export interface PatchOptions { readonly cached?: boolean; readonly check?: boolean; readonly env?: Record; + readonly threeWay?: boolean; readonly signal?: AbortSignal; } @@ -359,6 +360,7 @@ function buildApplyArgs(patchPath: string, options: PatchOptions): string[] { const args = ["apply"]; if (options.check) args.push("--check"); if (options.cached) args.push("--cached"); + if (options.threeWay) args.push("--3way"); args.push("--binary", patchPath); return args; } diff --git a/packages/coding-agent/test/task/isolation-runner.test.ts b/packages/coding-agent/test/task/isolation-runner.test.ts index 599883362..e8142c3f7 100644 --- a/packages/coding-agent/test/task/isolation-runner.test.ts +++ b/packages/coding-agent/test/task/isolation-runner.test.ts @@ -44,6 +44,25 @@ describe("mergeIsolatedChanges", () => { expect(outcome.summary).toContain("nested repository patches captured"); }); + it("surfaces branch preparation errors instead of reporting no changes", async () => { + const mergeSpy = vi.spyOn(worktreeModule, "mergeTaskBranches"); + const outcome = await mergeIsolatedChanges({ + repoRoot: "/repo", + mergeMode: "branch", + result: result({ + error: "Merge failed: git apply --3way failed for task dirty-context: conflict", + }), + }); + + expect(mergeSpy).not.toHaveBeenCalled(); + expect(outcome.changesApplied).toBe(false); + expect(outcome.hadAnyChanges).toBe(false); + expect(outcome.mergedBranchForNestedPatches).toBe(false); + expect(outcome.summary).toContain("Branch merge failed before a task branch could be created"); + expect(outcome.summary).toContain("git apply --3way failed"); + expect(outcome.summary).not.toContain("No changes to apply"); + }); + it("does not mark failed branch-mode runs as nested-patch eligible", async () => { const outcome = await mergeIsolatedChanges({ repoRoot: "/repo", diff --git a/packages/coding-agent/test/task/worktree.test.ts b/packages/coding-agent/test/task/worktree.test.ts index 3dcdf6d1b..fe3968bad 100644 --- a/packages/coding-agent/test/task/worktree.test.ts +++ b/packages/coding-agent/test/task/worktree.test.ts @@ -6,6 +6,8 @@ import { applyNestedPatches, captureBaseline, captureDeltaPatch, + cleanupTaskBranches, + commitToBranch, ensureIsolation, getGitNoIndexNullPath, getRepoRoot, @@ -233,6 +235,44 @@ describe("worktree isolation helpers", () => { expect(stashList).toBe(""); }); + it("commits isolated edits when parent dirt only changes nearby context", async () => { + const fixtureName = "EXP_DIRTY_TEST.txt"; + const fixturePath = path.join(repo, fixtureName); + const cleanLines = Array.from({ length: 10 }, (_, index) => `line${index + 1}`); + await fs.writeFile(fixturePath, `${cleanLines.join("\n")}\n`); + await runGit(repo, ["add", fixtureName]); + await runGit(repo, ["commit", "-q", "-m", "add dirty merge fixture"]); + + const parentDirtyLines = cleanLines.map((line, index) => (index === 1 ? "LINE2-DIRTY-PARENT" : line)); + await fs.writeFile(fixturePath, `${parentDirtyLines.join("\n")}\n`); + const baseline = await captureBaseline(repo); + + const isoRoot = await fs.mkdtemp(path.join(os.tmpdir(), "omp-worktree-iso-")); + tempDirs.push(isoRoot); + const iso = path.join(isoRoot, "repo"); + await runGit(isoRoot, ["clone", "-q", repo, iso]); + await runGit(iso, ["config", "user.email", "test@example.com"]); + await runGit(iso, ["config", "user.name", "Test User"]); + const isolatedLines = parentDirtyLines.map((line, index) => (index === 4 ? "LINE5-AGENT-EDIT" : line)); + await fs.writeFile(path.join(iso, fixtureName), `${isolatedLines.join("\n")}\n`); + + const taskId = `dirty-context-${path.basename(isoRoot)}`; + let branchName = `omp/task/${taskId}`; + try { + const commitResult = await commitToBranch(iso, baseline, taskId, "dirty context merge"); + if (!commitResult?.branchName) throw new Error("expected task branch"); + branchName = commitResult.branchName; + + const mergeResult = await mergeTaskBranches(repo, [{ branchName, taskId }]); + const finalContent = await fs.readFile(fixturePath, "utf8"); + + expect(mergeResult).toEqual({ failed: [], merged: [branchName] }); + expect(finalContent).toBe(`${isolatedLines.join("\n")}\n`); + } finally { + await cleanupTaskBranches(repo, [branchName]); + } + }); + it("subtracts baseline dirty state even when the task commits it", async () => { await Promise.all([ fs.writeFile(path.join(repo, "merged.txt"), "baseline dirty change\n"), From da715aae7cb7a3c49ce22a943fbf80c89cae0ec7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 00:13:07 +0000 Subject: [PATCH 13/88] fix(coding-agent): preserve agent commits across isolated branch merges MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When an isolated task agent commits its own changes before yielding, the harness used to collapse the captured delta into one AI-summarized commit and discard the agent's commit messages and authorship entirely. This violated commit discipline for agentic swarms — multiple logical commits ("fix bug" + "add test") became a single opaque commit, and the agent's commit object (which lived in isolation/.git/objects under overlayfs/rcopy) was lost when cleanupIsolation tore down the overlay. commitToBranch now detects when isolation HEAD moved past baseline.root .headCommit. When it has, the function git-fetches the agent's HEAD into the parent repo as omp/task/${taskId} so the commit objects survive cleanupIsolation, and stamps the captured baselineSha onto the returned CommitToBranchResult. mergeTaskBranches cherry-picks the inclusive range baseSha..branchName when baseSha is provided, replaying each agent commit verbatim with its original message and author. Any uncommitted leftover (staged, unstaged, untracked) on top of the agent's last commit becomes one trailing AI-summarized commit on the same branch. Falls back to the legacy single-commit path when the agent never moved HEAD (purely dirty working tree); existing patch-mode flow is untouched. Fixes #3842 --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/task/isolation-runner.ts | 8 +- packages/coding-agent/src/task/types.ts | 6 + packages/coding-agent/src/task/worktree.ts | 92 +++++++++--- .../coding-agent/test/task/worktree.test.ts | 135 ++++++++++++++++++ 5 files changed, 228 insertions(+), 17 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..a449971bb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Isolated branch-mode task merges now preserve the agent's own commits (message and author) instead of collapsing every diff into a single AI-summarized commit. `commitToBranch` detects when the subagent moved HEAD past the baseline, transfers the new commit objects into the parent repo via `git fetch`, and `mergeTaskBranches` cherry-picks the inclusive range `baseSha..omp/task/` so each commit replays verbatim; any uncommitted leftover on top of the agent's last commit lands as one trailing AI-summarized commit ([#3842](https://github.com/can1357/oh-my-pi/issues/3842)). + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/task/isolation-runner.ts b/packages/coding-agent/src/task/isolation-runner.ts index 455bba052..3a7b9ab68 100644 --- a/packages/coding-agent/src/task/isolation-runner.ts +++ b/packages/coding-agent/src/task/isolation-runner.ts @@ -154,6 +154,7 @@ export async function runIsolatedSubprocess(opts: IsolatedRunOptions): Promise { export interface CommitToBranchResult { branchName?: string; nestedPatches: NestedRepoPatch[]; + /** + * SHA of the parent-repo commit the task branch was created on top of, so + * {@link mergeTaskBranches} can cherry-pick the range `baseSha..branchName` + * and preserve every agent commit's message and author. + */ + baseSha?: string; } /** - * Commit task-only changes to a new branch. - * Only root repo changes go on the branch. Nested repo patches are returned - * separately since the parent git can't track files inside gitlinks. + * Capture task-only changes from the isolation worktree onto a parent-repo + * branch named `omp/task/${taskId}`. Only root-repo changes go on the branch; + * nested-repo patches are returned separately because the parent git can't + * track files inside gitlinks. + * + * If the agent committed its own changes inside isolation (HEAD moved past + * `baseline.root.headCommit`), this transfers those commit objects into the + * parent repo via `git fetch` and points the branch at the agent's HEAD — + * later cherry-pick of `baseSha..branchName` replays every commit with its + * original message and author preserved. Any uncommitted leftover (staged, + * unstaged, untracked) on top of the agent's last commit becomes one + * additional commit with an AI-generated message. + * + * If the agent did not commit, the captured delta is collapsed onto a single + * branch commit with an AI-generated (or fallback) message — the legacy + * behaviour. + * + * Returns `null` when no root or nested changes exist. */ export async function commitToBranch( isolationDir: string, @@ -474,6 +495,10 @@ export async function commitToBranch( description: string | undefined, commitMessage?: (diff: string) => Promise, ): Promise { + const baselineSha = baseline.root.headCommit; + const isolationHead = (await git.head.sha(isolationDir)) ?? ""; + const agentCommitted = isolationHead !== "" && isolationHead !== baselineSha; + const { rootPatch, nestedPatches } = await captureDeltaPatch(isolationDir, baseline); if (!rootPatch.trim() && nestedPatches.length === 0) return null; @@ -481,14 +506,40 @@ export async function commitToBranch( const branchName = `omp/task/${taskId}`; const fallbackMessage = description || taskId; - // Only create a branch if the root repo has changes - if (rootPatch.trim()) { + let branchCreated = false; + let leftoverPatch = ""; + + if (agentCommitted) { + // Transfer the agent's commit objects (which live in isolation's `.git`, + // stranded once `cleanupIsolation` tears the overlay down) into the parent + // repo's object DB and create the branch at the agent's HEAD. `+HEAD:…` + // force-overwrites a stale branch from a prior run. + await git.fetch(repoRoot, isolationDir, "HEAD", `refs/heads/${branchName}`); + branchCreated = true; + + // Leftover = anything still uncommitted in isolation on top of the + // agent's last commit (staged, unstaged, untracked). The agent didn't + // commit it, so it goes in as one AI-summarized trailing commit. + leftoverPatch = await captureRepoDeltaPatch(isolationDir, { + repoRoot: isolationDir, + headCommit: isolationHead, + staged: "", + unstaged: "", + untracked: [], + untrackedPatch: "", + }); + } else if (rootPatch.trim()) { await git.branch.create(repoRoot, branchName); + branchCreated = true; + leftoverPatch = rootPatch; + } + + if (branchCreated && leftoverPatch.trim()) { const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`); try { await git.worktree.add(repoRoot, tmpDir, branchName); try { - await git.patch.applyText(tmpDir, rootPatch); + await git.patch.applyText(tmpDir, leftoverPatch); } catch (err) { if (err instanceof git.GitCommandError) { const stderr = err.result.stderr.slice(0, 2000); @@ -496,15 +547,15 @@ export async function commitToBranch( taskId, exitCode: err.result.exitCode, stderr, - patchSize: rootPatch.length, - patchHead: rootPatch.slice(0, 500), + patchSize: leftoverPatch.length, + patchHead: leftoverPatch.slice(0, 500), }); throw new Error(`git apply failed for task ${taskId}: ${stderr}`); } throw err; } await git.stage.files(tmpDir); - const msg = (commitMessage && (await commitMessage(rootPatch))) || fallbackMessage; + const msg = (commitMessage && (await commitMessage(leftoverPatch))) || fallbackMessage; await git.commit(tmpDir, msg); } finally { await git.worktree.tryRemove(repoRoot, tmpDir); @@ -512,7 +563,11 @@ export async function commitToBranch( } } - return { branchName: rootPatch.trim() ? branchName : undefined, nestedPatches }; + return { + branchName: branchCreated ? branchName : undefined, + baseSha: baselineSha, + nestedPatches, + }; } export interface MergeBranchResult { @@ -524,13 +579,17 @@ export interface MergeBranchResult { } /** - * Cherry-pick task branch commits sequentially onto HEAD. - * Each branch has a single commit that gets replayed cleanly. - * Stops on first conflict and reports which branches succeeded. + * Cherry-pick task branch commits sequentially onto HEAD. When `baseSha` is + * provided the cherry-pick uses the inclusive range `baseSha..branchName`, + * replaying every commit individually and preserving each commit's message + * and author. When omitted, the branch is cherry-picked as a single commit + * (legacy callers). + * + * Stops on the first conflict and reports which branches succeeded. */ export async function mergeTaskBranches( repoRoot: string, - branches: Array<{ branchName: string; taskId: string; description?: string }>, + branches: Array<{ branchName: string; taskId: string; description?: string; baseSha?: string }>, ): Promise { // Serialize against other in-process git mutations on this repo: concurrent // background merges interleaving stash push/pop + cherry-pick would corrupt @@ -546,9 +605,10 @@ export async function mergeTaskBranches( let conflictResult: MergeBranchResult | undefined; try { - for (const { branchName } of branches) { + for (const { branchName, baseSha } of branches) { try { - await git.cherryPick(repoRoot, branchName); + const target = baseSha ? `${baseSha}..${branchName}` : branchName; + await git.cherryPick(repoRoot, target); } catch (err) { try { await git.cherryPick.abort(repoRoot); diff --git a/packages/coding-agent/test/task/worktree.test.ts b/packages/coding-agent/test/task/worktree.test.ts index 3dcdf6d1b..17d18dfae 100644 --- a/packages/coding-agent/test/task/worktree.test.ts +++ b/packages/coding-agent/test/task/worktree.test.ts @@ -6,6 +6,7 @@ import { applyNestedPatches, captureBaseline, captureDeltaPatch, + commitToBranch, ensureIsolation, getGitNoIndexNullPath, getRepoRoot, @@ -426,3 +427,137 @@ describe("applyNestedPatches", () => { expect(stashList).toContain("omp-isolation-"); }); }); + +describe("commitToBranch preserves agent commits", () => { + let parent: string; + let isolation: string; + + async function gitr(repo: string, args: string[]): Promise { + return runGit(repo, args); + } + + beforeEach(async () => { + parent = await fs.mkdtemp(path.join(os.tmpdir(), "omp-commit-parent-")); + isolation = await fs.mkdtemp(path.join(os.tmpdir(), "omp-commit-iso-")); + await gitr(parent, ["init", "-q", "-b", "main"]); + await gitr(parent, ["config", "user.email", "user@example.com"]); + await gitr(parent, ["config", "user.name", "Parent User"]); + await fs.writeFile( + path.join(parent, "EXP_CLEAN_COMMIT.txt"), + "line1\nline2\nline3\nline4\nline5\nline6\nline7\nline8\nline9\nline10\n", + ); + await gitr(parent, ["add", "."]); + await gitr(parent, ["commit", "-q", "-m", "add clean test fixture"]); + + // Simulate copy-on-write isolation: a real local clone so the agent's + // commit objects live in `isolation/.git`, just like the overlay/rcopy + // isolation backends would arrange them at runtime. + await fs.rm(isolation, { recursive: true, force: true }); + await gitr(parent, ["clone", "-q", "--no-hardlinks", "--local", parent, isolation]); + await gitr(isolation, ["config", "user.email", "agent@example.com"]); + await gitr(isolation, ["config", "user.name", "Agent User"]); + }); + + afterEach(async () => { + await Promise.all([removeWithRetries(parent), removeWithRetries(isolation)]); + }); + + // Reproduces issue #3842: agent commits with a specific message inside + // isolation; the merged commit on the parent branch must keep that exact + // message instead of an AI-generated summary. + it("preserves the agent's commit message after merge", async () => { + const baseline = await captureBaseline(parent); + + await fs.writeFile( + path.join(isolation, "EXP_CLEAN_COMMIT.txt"), + "line1\nline2\nline3\nline4\nLINE5-AGENT-WITH-MESSAGE\nline6\nline7\nline8\nline9\nline10\n", + ); + await gitr(isolation, ["add", "EXP_CLEAN_COMMIT.txt"]); + const agentMessage = "fix(test): agent committed with specific message for preservation check"; + await gitr(isolation, ["commit", "-q", "-m", agentMessage]); + + const taskId = "preservation-check"; + const aiMessage = vi.fn(async () => "fix: update line5 in clean commit example"); + const result = await commitToBranch(isolation, baseline, taskId, undefined, aiMessage); + + expect(result?.branchName).toBe(`omp/task/${taskId}`); + expect(result?.baseSha).toBe(baseline.root.headCommit); + // commitMessage callback must NOT have been invoked — the agent's + // message is taken verbatim. + expect(aiMessage).not.toHaveBeenCalled(); + + const branchSubject = await gitr(parent, ["log", "-1", "--pretty=%s", result!.branchName!]); + expect(branchSubject).toBe(agentMessage); + + const merge = await mergeTaskBranches(parent, [ + { branchName: result!.branchName!, taskId, baseSha: result!.baseSha! }, + ]); + expect(merge.failed).toEqual([]); + expect(merge.merged).toEqual([result!.branchName!]); + + const headSubject = await gitr(parent, ["log", "-1", "--pretty=%s"]); + expect(headSubject).toBe(agentMessage); + }); + + it("preserves every message when the agent makes multiple commits", async () => { + const baseline = await captureBaseline(parent); + + await fs.writeFile(path.join(isolation, "a.txt"), "alpha\n"); + await gitr(isolation, ["add", "a.txt"]); + await gitr(isolation, ["commit", "-q", "-m", "feat: add alpha file"]); + await fs.writeFile(path.join(isolation, "b.txt"), "beta\n"); + await gitr(isolation, ["add", "b.txt"]); + await gitr(isolation, ["commit", "-q", "-m", "test: add beta coverage"]); + + const result = await commitToBranch(isolation, baseline, "multi", undefined); + expect(result?.branchName).toBe("omp/task/multi"); + + const merge = await mergeTaskBranches(parent, [ + { branchName: result!.branchName!, taskId: "multi", baseSha: result!.baseSha! }, + ]); + expect(merge).toEqual({ failed: [], merged: ["omp/task/multi"] }); + + const subjects = (await gitr(parent, ["log", "-2", "--pretty=%s"])).split("\n"); + expect(subjects).toEqual(["test: add beta coverage", "feat: add alpha file"]); + }); + + it("appends one trailing commit when the agent leaves uncommitted work after committing", async () => { + const baseline = await captureBaseline(parent); + + await fs.writeFile(path.join(isolation, "a.txt"), "alpha\n"); + await gitr(isolation, ["add", "a.txt"]); + await gitr(isolation, ["commit", "-q", "-m", "feat: add alpha file"]); + // Uncommitted change on top of the agent's commit — should land as one + // extra commit with the AI-generated message, NOT silently dropped. + await fs.writeFile(path.join(isolation, "b.txt"), "beta\n"); + + const aiMessage = vi.fn(async () => "chore: leftover beta wip"); + const result = await commitToBranch(isolation, baseline, "leftover", undefined, aiMessage); + expect(result?.branchName).toBe("omp/task/leftover"); + expect(aiMessage).toHaveBeenCalledTimes(1); + + const subjects = (await gitr(parent, ["log", "-2", "--pretty=%s", result!.branchName!])).split("\n"); + expect(subjects).toEqual(["chore: leftover beta wip", "feat: add alpha file"]); + }); + + it("falls back to the AI-generated message when the agent never committed", async () => { + const baseline = await captureBaseline(parent); + + await fs.writeFile(path.join(isolation, "a.txt"), "alpha\n"); + + const aiMessage = vi.fn(async () => "feat: add alpha"); + const result = await commitToBranch(isolation, baseline, "nocommit", undefined, aiMessage); + + expect(result?.branchName).toBe("omp/task/nocommit"); + expect(aiMessage).toHaveBeenCalledTimes(1); + + const branchSubject = await gitr(parent, ["log", "-1", "--pretty=%s", result!.branchName!]); + expect(branchSubject).toBe("feat: add alpha"); + }); + + it("returns null when nothing changed in isolation", async () => { + const baseline = await captureBaseline(parent); + const result = await commitToBranch(isolation, baseline, "empty", undefined); + expect(result).toBeNull(); + }); +}); From d120ba6b7db8b62b2174ee0e5f2e8506c05c0052 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 00:28:27 +0000 Subject: [PATCH 14/88] fix(coding-agent): filter baseline wip from preserved agent commits Dirty isolated baselines can be accidentally committed by subagents that run git add -A. Fetching the raw isolation HEAD then cherry-picking the range would replay that baseline WIP into parent history. Add a dirty-baseline replay path that rewrites each agent commit against the captured baseline tree, preserving the agent commit message and author while excluding staged, unstaged, and untracked changes that existed before isolation started. Clean baselines still use the raw git fetch path, and nested-only changes keep returning patches without creating an empty root branch. Add a regression for baseline staged + untracked WIP committed by the agent, asserting the task branch contains only the agent file and parent WIP remains staged/untracked after merge. Fixes #3842 --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/task/worktree.ts | 198 +++++++++++++----- packages/coding-agent/src/utils/git.ts | 36 ++++ .../coding-agent/test/task/worktree.test.ts | 42 ++++ 4 files changed, 228 insertions(+), 50 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a449971bb..48b2416a9 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Isolated branch-mode task merges now preserve the agent's own commits (message and author) instead of collapsing every diff into a single AI-summarized commit. `commitToBranch` detects when the subagent moved HEAD past the baseline, transfers the new commit objects into the parent repo via `git fetch`, and `mergeTaskBranches` cherry-picks the inclusive range `baseSha..omp/task/` so each commit replays verbatim; any uncommitted leftover on top of the agent's last commit lands as one trailing AI-summarized commit ([#3842](https://github.com/can1357/oh-my-pi/issues/3842)). +- Isolated branch-mode task merges now preserve the agent's own commits (message and author) instead of collapsing every diff into a single AI-summarized commit. `commitToBranch` detects when the subagent moved HEAD past the baseline, transfers clean-baseline commit objects into the parent repo via `git fetch`, rewrites dirty-baseline commits against the captured baseline WIP so user staged/unstaged/untracked changes are filtered out, and `mergeTaskBranches` cherry-picks the inclusive range `baseSha..omp/task/` so each task commit replays with its original message; any uncommitted leftover on top of the agent's last commit lands as one trailing AI-summarized commit ([#3842](https://github.com/can1357/oh-my-pi/issues/3842)). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/src/task/worktree.ts b/packages/coding-agent/src/task/worktree.ts index dbd3049cf..5a67b1896 100644 --- a/packages/coding-agent/src/task/worktree.ts +++ b/packages/coding-agent/src/task/worktree.ts @@ -468,19 +468,115 @@ export interface CommitToBranchResult { baseSha?: string; } +function baselineHasRootWip(baseline: RepoBaseline): boolean { + return !!(baseline.staged.trim() || baseline.unstaged.trim() || baseline.untrackedPatch.trim()); +} + +async function commitPatchToBranchWorktree( + tmpDir: string, + taskId: string, + patchText: string, + message: string, + author?: git.CommitAuthor, +): Promise { + try { + await git.patch.applyText(tmpDir, patchText); + } catch (err) { + if (err instanceof git.GitCommandError) { + const stderr = err.result.stderr.slice(0, 2000); + logger.error("commitToBranch: git apply failed", { + taskId, + exitCode: err.result.exitCode, + stderr, + patchSize: patchText.length, + patchHead: patchText.slice(0, 500), + }); + throw new Error(`git apply failed for task ${taskId}: ${stderr}`); + } + throw err; + } + await git.stage.files(tmpDir); + await git.commit(tmpDir, message, author ? { author } : {}); +} + +interface FilteredAgentReplayOptions { + baseline: WorktreeBaseline; + branchName: string; + commitMessage?: (diff: string) => Promise; + fallbackMessage: string; + isolationDir: string; + isolationHead: string; + repoRoot: string; + rootPatch: string; + taskId: string; +} + +async function replayFilteredAgentCommits(opts: FilteredAgentReplayOptions): Promise { + const baselineSha = opts.baseline.root.headCommit; + await git.branch.create(opts.repoRoot, opts.branchName, baselineSha); + + const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`); + try { + await git.worktree.add(opts.repoRoot, tmpDir, opts.branchName); + const agentCommits = await git.revList.range(opts.isolationDir, baselineSha, opts.isolationHead); + const dirtyBaselineTree = await writeSyntheticTree(opts.isolationDir, baselineSha, [ + opts.baseline.root.staged, + opts.baseline.root.unstaged, + opts.baseline.root.untrackedPatch, + ]); + let previousFilteredTree = baselineSha; + + for (const commitSha of agentCommits) { + const taskStatePatch = await git.diff.tree(opts.isolationDir, dirtyBaselineTree, `${commitSha}^{tree}`, { + allowFailure: true, + binary: true, + }); + const currentFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [taskStatePatch]); + const commitPatch = await git.diff.tree(opts.repoRoot, previousFilteredTree, currentFilteredTree, { + allowFailure: true, + binary: true, + }); + if (commitPatch.trim()) { + const details = await git.commitDetails(opts.isolationDir, commitSha); + await commitPatchToBranchWorktree( + tmpDir, + opts.taskId, + commitPatch, + details.message || commitSha, + details.author, + ); + } + previousFilteredTree = currentFilteredTree; + } + + const finalFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [opts.rootPatch]); + const leftoverPatch = await git.diff.tree(opts.repoRoot, previousFilteredTree, finalFilteredTree, { + allowFailure: true, + binary: true, + }); + if (leftoverPatch.trim()) { + const msg = (opts.commitMessage && (await opts.commitMessage(leftoverPatch))) || opts.fallbackMessage; + await commitPatchToBranchWorktree(tmpDir, opts.taskId, leftoverPatch, msg); + } + } finally { + await git.worktree.tryRemove(opts.repoRoot, tmpDir); + await fs.rm(tmpDir, { recursive: true, force: true }); + } +} + /** * Capture task-only changes from the isolation worktree onto a parent-repo * branch named `omp/task/${taskId}`. Only root-repo changes go on the branch; * nested-repo patches are returned separately because the parent git can't * track files inside gitlinks. * - * If the agent committed its own changes inside isolation (HEAD moved past - * `baseline.root.headCommit`), this transfers those commit objects into the - * parent repo via `git fetch` and points the branch at the agent's HEAD — - * later cherry-pick of `baseSha..branchName` replays every commit with its - * original message and author preserved. Any uncommitted leftover (staged, - * unstaged, untracked) on top of the agent's last commit becomes one - * additional commit with an AI-generated message. + * If the agent committed inside isolation (HEAD moved past + * `baseline.root.headCommit`), clean-baseline runs fetch the raw commit range + * into the parent repo and later cherry-pick `baseSha..branchName`, preserving + * every message and author verbatim. Dirty-baseline runs rewrite each agent + * commit against the captured baseline WIP before committing it to the task + * branch, so user staged/unstaged/untracked changes present at isolation + * start are not replayed into the parent commit history. * * If the agent did not commit, the captured delta is collapsed onto a single * branch commit with an AI-generated (or fallback) message — the legacy @@ -501,62 +597,66 @@ export async function commitToBranch( const { rootPatch, nestedPatches } = await captureDeltaPatch(isolationDir, baseline); if (!rootPatch.trim() && nestedPatches.length === 0) return null; + if (!rootPatch.trim()) return { nestedPatches }; const repoRoot = baseline.root.repoRoot; const branchName = `omp/task/${taskId}`; const fallbackMessage = description || taskId; let branchCreated = false; - let leftoverPatch = ""; if (agentCommitted) { - // Transfer the agent's commit objects (which live in isolation's `.git`, - // stranded once `cleanupIsolation` tears the overlay down) into the parent - // repo's object DB and create the branch at the agent's HEAD. `+HEAD:…` - // force-overwrites a stale branch from a prior run. - await git.fetch(repoRoot, isolationDir, "HEAD", `refs/heads/${branchName}`); - branchCreated = true; + if (baselineHasRootWip(baseline.root)) { + await replayFilteredAgentCommits({ + baseline, + branchName, + commitMessage, + fallbackMessage, + isolationDir, + isolationHead, + repoRoot, + rootPatch, + taskId, + }); + } else { + // Transfer the agent's commit objects (which live in isolation's `.git`, + // stranded once `cleanupIsolation` tears the overlay down) into the parent + // repo's object DB and create the branch at the agent's HEAD. `+HEAD:…` + // force-overwrites a stale branch from a prior run. + await git.fetch(repoRoot, isolationDir, "HEAD", `refs/heads/${branchName}`); - // Leftover = anything still uncommitted in isolation on top of the - // agent's last commit (staged, unstaged, untracked). The agent didn't - // commit it, so it goes in as one AI-summarized trailing commit. - leftoverPatch = await captureRepoDeltaPatch(isolationDir, { - repoRoot: isolationDir, - headCommit: isolationHead, - staged: "", - unstaged: "", - untracked: [], - untrackedPatch: "", - }); + // Leftover = anything still uncommitted in isolation on top of the + // agent's last commit (staged, unstaged, untracked). The agent didn't + // commit it, so it goes in as one AI-summarized trailing commit. + const leftoverPatch = await captureRepoDeltaPatch(isolationDir, { + repoRoot: isolationDir, + headCommit: isolationHead, + staged: "", + unstaged: "", + untracked: [], + untrackedPatch: "", + }); + if (leftoverPatch.trim()) { + const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`); + try { + await git.worktree.add(repoRoot, tmpDir, branchName); + const msg = (commitMessage && (await commitMessage(leftoverPatch))) || fallbackMessage; + await commitPatchToBranchWorktree(tmpDir, taskId, leftoverPatch, msg); + } finally { + await git.worktree.tryRemove(repoRoot, tmpDir); + await fs.rm(tmpDir, { recursive: true, force: true }); + } + } + } + branchCreated = true; } else if (rootPatch.trim()) { - await git.branch.create(repoRoot, branchName); + await git.branch.create(repoRoot, branchName, baselineSha); branchCreated = true; - leftoverPatch = rootPatch; - } - - if (branchCreated && leftoverPatch.trim()) { const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`); try { await git.worktree.add(repoRoot, tmpDir, branchName); - try { - await git.patch.applyText(tmpDir, leftoverPatch); - } catch (err) { - if (err instanceof git.GitCommandError) { - const stderr = err.result.stderr.slice(0, 2000); - logger.error("commitToBranch: git apply failed", { - taskId, - exitCode: err.result.exitCode, - stderr, - patchSize: leftoverPatch.length, - patchHead: leftoverPatch.slice(0, 500), - }); - throw new Error(`git apply failed for task ${taskId}: ${stderr}`); - } - throw err; - } - await git.stage.files(tmpDir); - const msg = (commitMessage && (await commitMessage(leftoverPatch))) || fallbackMessage; - await git.commit(tmpDir, msg); + const msg = (commitMessage && (await commitMessage(rootPatch))) || fallbackMessage; + await commitPatchToBranchWorktree(tmpDir, taskId, rootPatch, msg); } finally { await git.worktree.tryRemove(repoRoot, tmpDir); await fs.rm(tmpDir, { recursive: true, force: true }); diff --git a/packages/coding-agent/src/utils/git.ts b/packages/coding-agent/src/utils/git.ts index 86afc8bc0..c8de9e67c 100644 --- a/packages/coding-agent/src/utils/git.ts +++ b/packages/coding-agent/src/utils/git.ts @@ -74,8 +74,20 @@ export interface StatusOptions { readonly z?: boolean; } +export interface CommitAuthor { + readonly date?: string; + readonly email: string; + readonly name: string; +} + +export interface CommitDetails { + readonly author: CommitAuthor; + readonly message: string; +} + export interface CommitOptions { readonly allowEmpty?: boolean; + readonly author?: CommitAuthor; readonly files?: readonly string[]; readonly signal?: AbortSignal; } @@ -1122,6 +1134,10 @@ export const stage = { /** Create a commit with the given message (passed via stdin). */ export async function commit(cwd: string, message: string, options: CommitOptions = {}): Promise { const args = ["commit", "-F", "-"]; + if (options.author) { + args.push(`--author=${options.author.name} <${options.author.email}>`); + if (options.author.date) args.push(`--date=${options.author.date}`); + } if (options.allowEmpty) args.push("--allow-empty"); if (options.files?.length) args.push("--", ...options.files); return runChecked(cwd, args, { signal: options.signal, stdin: message }); @@ -1195,6 +1211,19 @@ export const show = Object.assign( }, ); +/** Read commit message and author metadata for replay/rewrite flows. */ +export async function commitDetails(cwd: string, revision: string, signal?: AbortSignal): Promise { + const raw = await runText(cwd, ["show", "-s", "--format=%an%x00%ae%x00%aI%x00%B", revision], { + readOnly: true, + signal, + }); + const [name = "", email = "", date = "", ...messageParts] = raw.split("\0"); + return { + author: { date, email, name }, + message: messageParts.join("\0").replace(/\n$/, ""), + }; +} + // ════════════════════════════════════════════════════════════════════════════ // API: log // ════════════════════════════════════════════════════════════════════════════ @@ -1212,6 +1241,13 @@ export const log = { }, }; +export const revList = { + /** Commits in `base..head`, oldest first. */ + async range(cwd: string, base: string, head: string, signal?: AbortSignal): Promise { + return splitLines(await runText(cwd, ["rev-list", "--reverse", `${base}..${head}`], { readOnly: true, signal })); + }, +}; + // ════════════════════════════════════════════════════════════════════════════ // API: branch // ════════════════════════════════════════════════════════════════════════════ diff --git a/packages/coding-agent/test/task/worktree.test.ts b/packages/coding-agent/test/task/worktree.test.ts index 17d18dfae..260fafa98 100644 --- a/packages/coding-agent/test/task/worktree.test.ts +++ b/packages/coding-agent/test/task/worktree.test.ts @@ -540,6 +540,48 @@ describe("commitToBranch preserves agent commits", () => { expect(subjects).toEqual(["chore: leftover beta wip", "feat: add alpha file"]); }); + it("filters baseline WIP when the agent commits with git add -A", async () => { + await fs.writeFile(path.join(parent, "staged.txt"), "baseline staged wip\n"); + await gitr(parent, ["add", "staged.txt"]); + await fs.writeFile(path.join(parent, "user-wip.txt"), "baseline untracked wip\n"); + await fs.writeFile(path.join(isolation, "staged.txt"), "baseline staged wip\n"); + await gitr(isolation, ["add", "staged.txt"]); + await fs.writeFile(path.join(isolation, "user-wip.txt"), "baseline untracked wip\n"); + const baseline = await captureBaseline(parent); + + await fs.writeFile( + path.join(isolation, "EXP_CLEAN_COMMIT.txt"), + "line1\nline2\nline3\nline4\nLINE5-AGENT-WITH-MESSAGE\nline6\nline7\nline8\nline9\nline10\n", + ); + await gitr(isolation, ["add", "-A"]); + const agentMessage = "fix(test): preserve message without baseline wip"; + await gitr(isolation, ["commit", "-q", "-m", agentMessage]); + + const aiMessage = vi.fn(async () => "fix: generated fallback"); + const result = await commitToBranch(isolation, baseline, "dirty-baseline", undefined, aiMessage); + expect(result?.branchName).toBe("omp/task/dirty-baseline"); + expect(aiMessage).not.toHaveBeenCalled(); + + const branchFiles = (await gitr(parent, ["show", "--name-only", "--pretty=format:", result!.branchName!])) + .split("\n") + .filter(Boolean); + expect(branchFiles).toEqual(["EXP_CLEAN_COMMIT.txt"]); + + const merge = await mergeTaskBranches(parent, [ + { branchName: result!.branchName!, taskId: "dirty-baseline", baseSha: result!.baseSha! }, + ]); + expect(merge).toEqual({ failed: [], merged: ["omp/task/dirty-baseline"] }); + + const [headSubject, status, fixture] = await Promise.all([ + gitr(parent, ["log", "-1", "--pretty=%s"]), + gitr(parent, ["status", "--porcelain=v1"]), + fs.readFile(path.join(parent, "EXP_CLEAN_COMMIT.txt"), "utf8"), + ]); + expect(headSubject).toBe(agentMessage); + expect(status.split("\n").sort()).toEqual(["?? user-wip.txt", "A staged.txt"]); + expect(fixture).toContain("LINE5-AGENT-WITH-MESSAGE"); + }); + it("falls back to the AI-generated message when the agent never committed", async () => { const baseline = await captureBaseline(parent); From d8bf76dfd6fc3522a825681363d0e4f609eb72f5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 00:34:02 +0000 Subject: [PATCH 15/88] fix(coding-agent): skipped rebuilding previous session display context on different-session switch AgentSession.switchSession() eagerly called buildDisplaySessionContext() before setSessionFile, walking the previous session's branch and expanding every compaction entry's snapcompact archive and openaiRemoteCompaction replacementHistory into messages. For huge pre-fix sessions that materialized GBs of data and OOMed in-TUI /resume even after the streaming loader fix. The snapshot is only needed for same-session reloads, where #didSessionMessagesChange compares the pre/post message arrays to detect rollback edits. Different-session switches skip the call entirely; the error-recovery path rebuilds the previous context on demand from the restored state so MCP-selection restoration still has its inputs. Added a regression test (test/agent-session-switch-prev-context.test.ts) that spies on sessionManager.buildSessionContext across switchSession and asserts the expected call count and target file per branch. Fixes #3846 --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/session/agent-session.ts | 19 ++- .../agent-session-switch-prev-context.test.ts | 161 ++++++++++++++++++ 3 files changed, 181 insertions(+), 3 deletions(-) create mode 100644 packages/coding-agent/test/agent-session-switch-prev-context.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c68409f79..c1d43ea68 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed in-TUI `/resume` materializing the previous session's display context (snapcompact archives + OpenAI Responses replay payloads) before switching files, which could exhaust memory on huge pre-fix sessions. `AgentSession.switchSession` now only snapshots the prior context for same-session reloads, where it is needed for rollback comparison. ([#3846](https://github.com/can1357/oh-my-pi/issues/3846)) + ## [16.2.6] - 2026-06-29 ### Changed diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 049d8ac03..bc23e2dd7 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -13150,7 +13150,15 @@ export class AgentSession { // Flush pending writes before switching so restore snapshots reflect committed state. await this.sessionManager.flush(); const previousSessionState = this.sessionManager.captureState(); - const previousSessionContext = this.buildDisplaySessionContext(); + // Only same-session reloads compare against the prior context to detect + // rollback edits (`#didSessionMessagesChange` below). Building it for a + // different-session switch is a pure waste — and on huge pre-fix sessions + // it materializes every persisted snapcompact frame plus the + // `openaiRemoteCompaction.replacementHistory` payload into messages, + // blowing the heap before the new session even loads (issue #3846). The + // error-recovery path rebuilds the context on demand from the restored + // state instead. + const previousSessionContext = switchingToDifferentSession ? undefined : this.buildDisplaySessionContext(); // switchSession replaces these arrays wholesale during load/rollback, so retaining // the existing message objects is sufficient and avoids structured-clone failures for // extension/custom metadata that is valid to persist but not cloneable. @@ -13189,7 +13197,7 @@ export class AgentSession { const sessionContext = this.buildDisplaySessionContext(); const didReloadConversationChange = - !switchingToDifferentSession && + previousSessionContext !== undefined && this.#didSessionMessagesChange(previousSessionContext.messages, sessionContext.messages); const fallbackSelectedMCPToolNames = this.#getSessionDefaultSelectedMCPToolNames(sessionPath); await this.#restoreMCPSelectionsForSessionContext(sessionContext, { fallbackSelectedMCPToolNames }); @@ -13305,7 +13313,12 @@ export class AgentSession { this.#rekeyMnemopiMemoryForCurrentSessionId(); let restoreMcpError: unknown; try { - await this.#restoreMCPSelectionsForSessionContext(previousSessionContext, { + // `previousSessionContext` was skipped on different-session switches to + // avoid materializing the previous session's heavy compaction payload + // in the success path; rebuild it here on demand from the restored + // state so MCP selection restoration still has its inputs. + const mcpRestoreContext = previousSessionContext ?? this.buildDisplaySessionContext(); + await this.#restoreMCPSelectionsForSessionContext(mcpRestoreContext, { fallbackSelectedMCPToolNames: previousFallbackSelectedMCPToolNames, }); } catch (mcpError) { diff --git a/packages/coding-agent/test/agent-session-switch-prev-context.test.ts b/packages/coding-agent/test/agent-session-switch-prev-context.test.ts new file mode 100644 index 000000000..4fa599d8e --- /dev/null +++ b/packages/coding-agent/test/agent-session-switch-prev-context.test.ts @@ -0,0 +1,161 @@ +import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test"; +import * as path from "node:path"; +import { Agent } from "@oh-my-pi/pi-agent-core"; +import type { Model } from "@oh-my-pi/pi-ai"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import type { BuildSessionContextOptions, SessionContext } from "@oh-my-pi/pi-coding-agent/session/session-context"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +/** + * Regression for issue #3846: in-TUI `/resume` rebuilt the *previous* + * session's display context before switching files. That call expands persisted + * snapcompact archives and `openaiRemoteCompaction.replacementHistory` payloads + * into messages, which can OOM on huge pre-fix sessions even though the loader + * itself streams. The previous context is only needed for same-session reloads + * (where `#didSessionMessagesChange` compares against the freshly rebuilt one); + * different-session switches MUST skip that work. + */ +describe("AgentSession.switchSession previous-context build", () => { + let sharedDir: TempDir; + let authStorage: AuthStorage; + let modelRegistry: ModelRegistry; + let model: Model; + const tempDirs: TempDir[] = []; + const sessions: AgentSession[] = []; + + beforeAll(async () => { + sharedDir = TempDir.createSync("@pi-switch-prev-ctx-shared-"); + authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + modelRegistry = new ModelRegistry(authStorage); + const bundled = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!bundled) throw new Error("Expected built-in anthropic model to exist"); + model = bundled; + }); + + afterAll(async () => { + authStorage.close(); + try { + await sharedDir.remove(); + } catch {} + }); + + afterEach(async () => { + while (sessions.length > 0) { + await sessions.pop()?.dispose(); + } + for (const dir of tempDirs.splice(0)) { + try { + await dir.remove(); + } catch {} + } + }); + + function buildSession(tempDir: TempDir): { session: AgentSession; sessionManager: SessionManager } { + const sessionManager = SessionManager.create(tempDir.path(), tempDir.path()); + const agent = new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }); + const session = new AgentSession({ + agent, + sessionManager, + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry, + }); + sessions.push(session); + return { session, sessionManager }; + } + + /** Wrap `sessionManager.buildSessionContext` so each call's caller-visible + * state (the manager's currently-loaded session file) is recorded in + * invocation order. The constructor itself calls `buildSessionContext` + * once; spying *after* construction means only switchSession-driven calls + * are observed. */ + function instrumentBuildSessionContext(sessionManager: SessionManager): { + calls: Array<{ sessionFile: string | undefined; transcript: boolean | undefined }>; + restore: () => void; + } { + const calls: Array<{ sessionFile: string | undefined; transcript: boolean | undefined }> = []; + const original = sessionManager.buildSessionContext.bind(sessionManager); + const patched = (options?: BuildSessionContextOptions): SessionContext => { + calls.push({ sessionFile: sessionManager.getSessionFile(), transcript: options?.transcript }); + return original(options); + }; + sessionManager.buildSessionContext = patched as SessionManager["buildSessionContext"]; + return { + calls, + restore: () => { + sessionManager.buildSessionContext = original; + }, + }; + } + + it("skips building the previous display context when switching to a different session", async () => { + const tempDir = TempDir.createSync("@pi-switch-prev-ctx-different-"); + tempDirs.push(tempDir); + + const { session, sessionManager } = buildSession(tempDir); + sessionManager.appendMessage({ role: "user", content: "previous", timestamp: 1 }); + await sessionManager.flush(); + const previousSessionFile = sessionManager.getSessionFile(); + expect(previousSessionFile).toBeString(); + + const otherManager = SessionManager.create(tempDir.path(), tempDir.path()); + otherManager.appendMessage({ role: "user", content: "target", timestamp: 2 }); + await otherManager.flush(); + const targetSessionFile = otherManager.getSessionFile(); + expect(targetSessionFile).toBeString(); + expect(targetSessionFile).not.toBe(previousSessionFile); + await otherManager.close(); + + const { calls, restore } = instrumentBuildSessionContext(sessionManager); + try { + const switched = await session.switchSession(targetSessionFile!); + expect(switched).toBe(true); + expect(session.sessionFile).toBe(targetSessionFile); + } finally { + restore(); + } + + // The previous session's display context MUST NOT be materialized. Only + // the new target context (post-`setSessionFile`) should be built. + expect(calls).toEqual([{ sessionFile: targetSessionFile!, transcript: undefined }]); + }); + + it("builds the previous display context for same-session reloads", async () => { + const tempDir = TempDir.createSync("@pi-switch-prev-ctx-reload-"); + tempDirs.push(tempDir); + + const { session, sessionManager } = buildSession(tempDir); + sessionManager.appendMessage({ role: "user", content: "current", timestamp: 1 }); + await sessionManager.flush(); + const sessionFile = sessionManager.getSessionFile(); + expect(sessionFile).toBeString(); + + const { calls, restore } = instrumentBuildSessionContext(sessionManager); + try { + const switched = await session.switchSession(sessionFile!); + expect(switched).toBe(true); + expect(session.sessionFile).toBe(sessionFile); + } finally { + restore(); + } + + // Same-session reload must snapshot the pre-reload context so + // `#didSessionMessagesChange` can detect rollback edits. + expect(calls).toEqual([ + { sessionFile: sessionFile!, transcript: undefined }, + { sessionFile: sessionFile!, transcript: undefined }, + ]); + }); +}); From a7efaf4a53b8b0d0885d7f558841e1710bb40182 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 02:52:57 +0200 Subject: [PATCH 16/88] test(cli): cover tiny model text download errors --- .../coding-agent/test/tiny-models-cli.test.ts | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/packages/coding-agent/test/tiny-models-cli.test.ts b/packages/coding-agent/test/tiny-models-cli.test.ts index 363d1eeba..f3acdd520 100644 --- a/packages/coding-agent/test/tiny-models-cli.test.ts +++ b/packages/coding-agent/test/tiny-models-cli.test.ts @@ -50,4 +50,29 @@ describe("tiny-models download model resolution", () => { results: [{ model: "lfm2-700m", ok: false, error: "Error: runtime install failed\n at worker" }], }); }); + + it("includes worker error details in text failures", async () => { + const output: string[] = []; + const isTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "isTTY"); + Object.defineProperty(process.stdout, "isTTY", { configurable: true, value: false }); + spyOn(process.stdout, "write").mockImplementation((chunk: string | Uint8Array) => { + output.push(typeof chunk === "string" ? chunk : new TextDecoder().decode(chunk)); + return true; + }); + spyOn(tinyTitleClient, "downloadModel").mockResolvedValue({ + ok: false, + error: "Error: runtime install failed\n at worker", + }); + + try { + await expect( + runTinyModelsCommand({ action: "download", model: "lfm2-700m", flags: {} }), + ).rejects.toThrow("One or more tiny title models failed to download"); + } finally { + if (isTtyDescriptor) Object.defineProperty(process.stdout, "isTTY", isTtyDescriptor); + else delete (process.stdout as typeof process.stdout & { isTTY?: boolean }).isTTY; + } + + expect(output.join("")).toContain("Failed to download LFM2 700M: runtime install failed."); + }); }); From b5737a844805e814b329833a9b351dcc21597485 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 00:59:29 +0000 Subject: [PATCH 17/88] fix(prompting): stopped waiting on timed-out gpu stdout Stopped the GPU probe from awaiting stdout EOF after the probe process exits non-zero, which can hang when a wrapper leaves descendants holding stdout open. The regression now covers a killed wrapper with an inherited stdout holder and verifies the scenario process exits before the deadline. --- .../coding-agent/src/system-prompt.test.ts | 18 ++++++++++--- packages/coding-agent/src/system-prompt.ts | 25 ++++++++++++++++--- 2 files changed, 35 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index 95560f240..a55f77208 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -10,7 +10,11 @@ interface ProbeRunResult { count: number; } -async function runProbeScenario(options: { runs: number; sleepSeconds?: number }): Promise { +async function runProbeScenario(options: { + runs: number; + sleepSeconds?: number; + holdStdoutOpen?: boolean; +}): Promise { const tempRoot = await fs.mkdtemp(path.join(os.tmpdir(), "omp-gpu-probe-")); try { const binDir = path.join(tempRoot, "bin"); @@ -21,7 +25,7 @@ async function runProbeScenario(options: { runs: number; sleepSeconds?: number } const lspciPath = path.join(binDir, "lspci"); await Bun.write( lspciPath, - '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', + '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ "$OMP_GPU_PROBE_HOLD_STDOUT_OPEN" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & wait "$!"; fi\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', ); await fs.chmod(lspciPath, 0o755); @@ -74,6 +78,11 @@ console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt } else { env.OMP_GPU_PROBE_SLEEP = String(options.sleepSeconds); } + if (options.holdStdoutOpen) { + env.OMP_GPU_PROBE_HOLD_STDOUT_OPEN = "true"; + } else { + delete env.OMP_GPU_PROBE_HOLD_STDOUT_OPEN; + } const childStartedAt = performance.now(); const child = Bun.spawn([process.execPath, scenarioPath], { stdout: "pipe", stderr: "pipe", env }); @@ -101,11 +110,12 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { }, 15_000); it("kills the GPU probe at the prep deadline", async () => { - const result = await runProbeScenario({ runs: 1, sleepSeconds: 7 }); + const result = await runProbeScenario({ runs: 1, sleepSeconds: 7, holdStdoutOpen: true }); + expect(result.cached).toEqual({ gpu: null }); expect(result.elapsedMs).toBeLessThan(6500); // Codex#3838: the child process MUST exit shortly after the deadline, - // not linger until the underlying probe (sleep 7) finishes on its own. + // not linger until a descendant holding stdout (sleep 7) exits on its own. expect(result.childElapsedMs).toBeLessThan(6500); }, 15_000); }); diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index c5b959a0c..d409d0e3a 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -111,8 +111,8 @@ function parseWmicTable(output: string, header: string): string | null { } const SYSTEM_PROMPT_PREP_TIMEOUT_MS = 5000; -/** Killed by the OS at this deadline, so a wedged probe cannot outlive the prep race. */ -const GPU_PROBE_TIMEOUT_MS = SYSTEM_PROMPT_PREP_TIMEOUT_MS; +/** Kept below prep timeout so timed-out probes can still write the null cache before fallback. */ +const GPU_PROBE_TIMEOUT_MS = SYSTEM_PROMPT_PREP_TIMEOUT_MS - 500; async function runGpuProbe(cmd: string[]): Promise { try { @@ -126,8 +126,25 @@ async function runGpuProbe(cmd: string[]): Promise { // dies at the deadline and lets getCachedGpu reach the null-cache write. killSignal: "SIGKILL", }); - const [stdout, exitCode] = await Promise.all([new Response(proc.stdout).text(), proc.exited]); - return exitCode === 0 ? stdout : null; + const stdoutReader = proc.stdout.getReader(); + let stdout = ""; + const decoder = new TextDecoder(); + const stdoutDone = (async () => { + while (true) { + const chunk = await stdoutReader.read(); + if (chunk.done) break; + stdout += decoder.decode(chunk.value, { stream: true }); + } + stdout += decoder.decode(); + })(); + const exitCode = await proc.exited; + if (exitCode !== 0) { + await stdoutReader.cancel().catch(() => undefined); + await stdoutDone.catch(() => undefined); + return null; + } + await stdoutDone; + return stdout; } catch { return null; } From e8090bb48ae228d306647b9abf466e40a4d70f68 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 02:58:47 +0200 Subject: [PATCH 18/88] feat: introduced binary file detection to prevent encoding corruption - Introduced `isProbablyBinary` utility to sniff file headers for NUL bytes or invalid UTF-8 sequences. - Updated `ReadTool` to use the binary sniffer, preventing mojibake corruption in output when reading non-text files. - Refined `file-mentions` auto-reads to skip binary files and mark them as `binary` in the message transcript. - Added comprehensive unit tests for binary detection logic, covering NUL bytes, truncated multibyte characters, and path-based file sniffing. --- packages/coding-agent/CHANGELOG.md | 3 + .../modes/utils/transcript-render-helpers.ts | 4 +- packages/coding-agent/src/session/messages.ts | 2 +- packages/coding-agent/src/tools/read.ts | 56 ++++++++--------- .../coding-agent/src/utils/file-mentions.ts | 11 +++- .../coding-agent/test/file-mentions.test.ts | 22 +++++++ packages/coding-agent/test/tools.test.ts | 26 +++++--- packages/utils/CHANGELOG.md | 4 ++ packages/utils/src/binary.ts | 50 +++++++++++++++ packages/utils/src/index.ts | 1 + packages/utils/test/binary.test.ts | 61 +++++++++++++++++++ 11 files changed, 201 insertions(+), 39 deletions(-) create mode 100644 packages/utils/src/binary.ts create mode 100644 packages/utils/test/binary.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 122258be0..a3f375afd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,9 @@ ### Changed +- Improved binary file detection to prevent terminal corruption from non-UTF8 content +- Updated file mention summaries to explicitly note skipped binary files + - Enabled contextual snapcompact shape resolution based on rendered text content ### Fixed diff --git a/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts b/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts index d66a6a030..74a70cd23 100644 --- a/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts +++ b/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts @@ -99,9 +99,9 @@ export function buildFileMentionBlock(files: FileMentionMessage["files"], indent const block = new TranscriptBlock(); for (const file of files) { let suffix: string; - if (file.skippedReason === "tooLarge") { + if (file.skippedReason === "tooLarge" || file.skippedReason === "binary") { const size = typeof file.byteSize === "number" ? formatBytes(file.byteSize) : "unknown size"; - suffix = `(skipped: ${size})`; + suffix = file.skippedReason === "binary" ? `(skipped: binary, ${size})` : `(skipped: ${size})`; } else { suffix = file.image ? "(image)" diff --git a/packages/coding-agent/src/session/messages.ts b/packages/coding-agent/src/session/messages.ts index b5eb34f2b..f72c829f5 100644 --- a/packages/coding-agent/src/session/messages.ts +++ b/packages/coding-agent/src/session/messages.ts @@ -488,7 +488,7 @@ export interface FileMentionMessage { /** File size in bytes, if known. */ byteSize?: number; /** Why the file contents were omitted from auto-read. */ - skippedReason?: "tooLarge"; + skippedReason?: "tooLarge" | "binary"; image?: ImageContent; }>; timestamp: number; diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 429fb9ff5..e57369238 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -14,7 +14,15 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; import { glob, type SummaryResult, summarizeCode } from "@oh-my-pi/pi-natives"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; -import { getRemoteDir, type ImageMetadata, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils"; +import { + getRemoteDir, + type ImageMetadata, + isProbablyBinary, + logger, + prompt, + readImageMetadata, + untilAborted, +} from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import { LRUCache } from "lru-cache/raw"; import { @@ -2314,6 +2322,25 @@ export class ReadTool implements AgentTool { content = [{ type: "text", text: `[Cannot read ${ext} file: conversion failed]` }]; } } else { + // Binary sniff before any UTF-8 text materialization. A binary file + // (font, object, archive, packed blob) decodes to NUL/control bytes and + // U+FFFD mojibake that corrupts the terminal and burns context. Images, + // notebooks, and markit-convertible documents were already routed above; + // everything reaching here is meant to be plain text. `:raw` stays the + // explicit escape hatch for reading bytes verbatim. This single guard + // covers both the multi-range and single-range disk paths below. + if (!isRawSelector(parsed) && (await isProbablyBinary(absolutePath))) { + return toolResult({ resolvedPath: absolutePath, suffixResolution }) + .text( + prependSuffixResolutionNotice( + `[Cannot read binary file '${formatPathRelativeToCwd(absolutePath, this.session.cwd)}' (${formatBytes(fileSize)}); not valid UTF-8 text. Use ':raw' to read bytes verbatim.]`, + suffixResolution, + ), + ) + .sourcePath(absolutePath) + .done(); + } + if ( parsed.kind === "none" && this.session.settings.get("read.summarize.enabled") && @@ -2449,33 +2476,6 @@ export class ReadTool implements AgentTool { // counts in `truncation` keep reflecting the source, not the trimmed // view — column truncation surfaces separately via `.limits()`. const rawSelector = isRawSelector(parsed); - // Binary sniff: NUL bytes mean the file is not displayable text - // (binary, or UTF-16 which has NULs in the ASCII range) — emit a - // notice instead of mojibake filling the line budget. `:raw` - // stays an explicit escape hatch. - // - // `collectedLines` covers the common case where at least one - // physical line terminates within the byte budget. Binary blobs - // without newlines (videos, archives, packed JSON) leave it - // empty; their bytes only land in `firstLinePreview`, which the - // `firstLineExceedsLimit` branch below would otherwise emit - // verbatim. Sniffing the preview here keeps the refusal uniform. - if (!rawSelector) { - const hasNul = (text: string): boolean => text.includes("\u0000"); - const binaryDetected = - collectedLines.some(hasNul) || (firstLinePreview !== undefined && hasNul(firstLinePreview.text)); - if (binaryDetected) { - return toolResult({ resolvedPath: absolutePath, suffixResolution }) - .text( - prependSuffixResolutionNotice( - `[Cannot read binary file '${formatPathRelativeToCwd(absolutePath, this.session.cwd)}' (${formatBytes(fileSize)}); content contains NUL bytes (binary or UTF-16 encoded)]`, - suffixResolution, - ), - ) - .sourcePath(absolutePath) - .done(); - } - } const maxColumns = resolveOutputMaxColumns(this.session.settings); // Column truncation is display-only. `collectedLines` MUST stay // byte-for-byte with the on-disk content so the snapshot recorded diff --git a/packages/coding-agent/src/utils/file-mentions.ts b/packages/coding-agent/src/utils/file-mentions.ts index 87ad2bfa1..b7e5a0d19 100644 --- a/packages/coding-agent/src/utils/file-mentions.ts +++ b/packages/coding-agent/src/utils/file-mentions.ts @@ -10,7 +10,7 @@ import path from "node:path"; import { formatHashlineHeader, formatNumberedLines, type SnapshotStore } from "@oh-my-pi/hashline"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { ImageContent } from "@oh-my-pi/pi-ai"; -import { formatAge, formatBytes, readImageMetadata } from "@oh-my-pi/pi-utils"; +import { formatAge, formatBytes, isProbablyBinary, readImageMetadata } from "@oh-my-pi/pi-utils"; import { canonicalSnapshotKey } from "../edit/file-snapshot-store"; import { normalizeToLF } from "../edit/normalize"; import type { FileMentionMessage } from "../session/messages"; @@ -257,6 +257,15 @@ export async function generateFileMentionMessages( }); continue; } + if (await isProbablyBinary(absolutePath)) { + files.push({ + path: resolvedPath, + content: `(skipped auto-read: binary file, ${formatBytes(stat.size)})`, + byteSize: stat.size, + skippedReason: "binary", + }); + continue; + } const content = await Bun.file(absolutePath).text(); const snapshotStore = options?.useHashLines ? options.snapshotStore : undefined; diff --git a/packages/coding-agent/test/file-mentions.test.ts b/packages/coding-agent/test/file-mentions.test.ts index a49b56c5d..3de5a1525 100644 --- a/packages/coding-agent/test/file-mentions.test.ts +++ b/packages/coding-agent/test/file-mentions.test.ts @@ -98,4 +98,26 @@ describe("generateFileMentionMessages path resolution", () => { expect(message.files).toHaveLength(1); expect(message.files[0]?.path).toBe("My Folder/my file.png"); }); + + test("skips auto-reading a binary file instead of injecting raw bytes", async () => { + const cwd = await createTempDir(); + // TTF header begins with a NUL run; auto-reading it as text would leak + // control bytes into the conversation (the reported bug). + await Bun.write(path.join(cwd, "Silver.ttf"), Buffer.from([0x00, 0x01, 0x00, 0x00, 0x00, 0x0c, 0x4f, 0x53])); + // A non-NUL invalid-UTF8 blob must be refused too, not just NUL-bearing files. + await Bun.write(path.join(cwd, "blob.bin"), Buffer.from([0x4d, 0x5a, 0xff, 0xfe, 0xc0, 0xc0])); + + const messages = await generateFileMentionMessages(["Silver.ttf", "blob.bin"], cwd); + expect(messages).toHaveLength(1); + const message = messages[0]; + if (message?.role !== "fileMention") { + throw new Error("expected file mention message"); + } + expect(message.files).toHaveLength(2); + for (const file of message.files) { + expect(file.skippedReason).toBe("binary"); + expect(file.content).toContain("binary file"); + expect(file.content).not.toContain("\u0000"); + } + }); }); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index 94a4e8eac..e9e222ce8 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -578,15 +578,27 @@ describe("Coding Agent Tools", () => { expect(output).toContain("Use :1 to read from the start, or :3 to read the last line."); }); - it("should emit a binary notice instead of mojibake for files with NUL bytes", async () => { - const testFile = path.join(testDir, "blob.bin"); - fs.writeFileSync(testFile, Buffer.from([0x61, 0x62, 0x63, 0x00, 0xff, 0xfe, 0x64, 0x65])); + it("should refuse binary files (NUL or invalid UTF-8) instead of emitting mojibake", async () => { + const nulFile = path.join(testDir, "blob.bin"); + fs.writeFileSync(nulFile, Buffer.from([0x61, 0x62, 0x63, 0x00, 0xff, 0xfe, 0x64, 0x65])); + // A header with no NUL but invalid UTF-8 (lone 0xFF/0xC0) must also refuse. + const invalidUtf8File = path.join(testDir, "font.ttfish"); + fs.writeFileSync(invalidUtf8File, Buffer.from([0x4d, 0x5a, 0xff, 0xfe, 0xc0, 0xc0, 0x90, 0x91])); - const result = await readTool.execute("test-call-binary-nul", { path: testFile }); - const output = getTextOutput(result); + for (const file of [nulFile, invalidUtf8File]) { + const output = getTextOutput(await readTool.execute("test-call-binary", { path: file })); + expect(output).toContain("Cannot read binary file"); + expect(output).not.toContain("\u0000"); + expect(output).not.toContain("\uFFFD"); + } + }); - expect(output).toContain("Cannot read binary file"); - expect(output).toContain("NUL bytes"); + it("reads a binary file verbatim when :raw is requested", async () => { + const testFile = path.join(testDir, "raw-blob.bin"); + fs.writeFileSync(testFile, Buffer.from([0x61, 0x62, 0x63, 0x00, 0x64, 0x65])); + + const output = getTextOutput(await readTool.execute("test-call-binary-raw", { path: `${testFile}:raw` })); + expect(output).not.toContain("Cannot read binary file"); }); it("should reject malformed internal-URL selectors instead of dumping the whole resource", async () => { diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 43275e441..9b6dd0d44 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added utility to detect binary files based on content sniffing + ## [16.2.6] - 2026-06-29 ### Added diff --git a/packages/utils/src/binary.ts b/packages/utils/src/binary.ts new file mode 100644 index 000000000..1245bd062 --- /dev/null +++ b/packages/utils/src/binary.ts @@ -0,0 +1,50 @@ +/** + * Content-based binary/text classification for files that are about to be + * decoded as UTF-8 text and shown to a model or user. + * + * The read tool and `@file` auto-read both materialize file bytes as UTF-8 + * strings. For a binary file (font, object, archive, packed blob) that decode + * is lossy: NUL bytes and invalid sequences survive as control characters and + * U+FFFD replacements, which corrupt terminal rendering and waste the context + * window with mojibake. Sniff the header first and refuse instead. + * + * @example + * if (await isProbablyBinary(path)) return "[binary file omitted]"; + * const text = await Bun.file(path).text(); + */ +import { peekFile, peekFileSync } from "./peek-file"; + +/** Header window sniffed for the binary heuristic; mirrors git's 8000-byte scan. */ +const BINARY_SNIFF_BYTES = 8192; + +/** + * Classify an in-memory byte header as binary (non-UTF-8-text). + * + * Binary when the header contains a NUL byte (true binary, plus UTF-16/UTF-32 + * text whose ASCII range is NUL-padded) or when it is not valid UTF-8. The + * decode runs in streaming mode so a multibyte sequence truncated at the header + * boundary is tolerated, while any genuinely invalid byte still fails — matching + * the strict `fatal` decode the `local://`/`ssh://` read paths already use. + */ +export function isProbablyBinaryHeader(header: Uint8Array): boolean { + if (header.indexOf(0) !== -1) return true; + try { + new TextDecoder("utf-8", { fatal: true }).decode(header, { stream: true }); + return false; + } catch { + return true; + } +} + +/** + * Sniff the first {@link BINARY_SNIFF_BYTES} of `filePath` and report whether it + * is binary (non-UTF-8-text). See {@link isProbablyBinaryHeader} for the rule. + */ +export function isProbablyBinary(filePath: string, maxBytes = BINARY_SNIFF_BYTES): Promise { + return peekFile(filePath, maxBytes, isProbablyBinaryHeader); +} + +/** Synchronous {@link isProbablyBinary}. */ +export function isProbablyBinarySync(filePath: string, maxBytes = BINARY_SNIFF_BYTES): boolean { + return peekFileSync(filePath, maxBytes, isProbablyBinaryHeader); +} diff --git a/packages/utils/src/index.ts b/packages/utils/src/index.ts index 23dc2bc35..3fcfe829e 100644 --- a/packages/utils/src/index.ts +++ b/packages/utils/src/index.ts @@ -1,5 +1,6 @@ export { once, untilAborted } from "./abortable"; export * from "./async"; +export * from "./binary"; export * from "./color"; export * from "./dirs"; export * from "./env"; diff --git a/packages/utils/test/binary.test.ts b/packages/utils/test/binary.test.ts new file mode 100644 index 000000000..de204b26f --- /dev/null +++ b/packages/utils/test/binary.test.ts @@ -0,0 +1,61 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { isProbablyBinary, isProbablyBinaryHeader, isProbablyBinarySync } from "@oh-my-pi/pi-utils/binary"; + +describe("isProbablyBinaryHeader", () => { + it("treats empty input as text", () => { + expect(isProbablyBinaryHeader(new Uint8Array(0))).toBe(false); + }); + + it("flags a NUL byte as binary", () => { + // TTF/OTF, WASM, ELF, UTF-16 text all carry NUL in their first bytes. + expect(isProbablyBinaryHeader(Buffer.from([0x00, 0x01, 0x00, 0x00]))).toBe(true); + }); + + it("flags invalid UTF-8 without a NUL as binary", () => { + // 0xFF/0xFE never appear in valid UTF-8; a font/object header with no + // early NUL still fails the fatal decode. + expect(isProbablyBinaryHeader(Buffer.from([0x4d, 0x5a, 0xff, 0xfe, 0xc0, 0xc0]))).toBe(true); + }); + + it("accepts plain ASCII text", () => { + expect(isProbablyBinaryHeader(Buffer.from("export const x = 1;\n", "utf-8"))).toBe(false); + }); + + it("accepts multibyte UTF-8 text", () => { + expect(isProbablyBinaryHeader(Buffer.from("héllo — 日本語 🚀\n", "utf-8"))).toBe(false); + }); + + it("tolerates a multibyte sequence truncated at the header boundary", () => { + // "😀" is 4 bytes (F0 9F 98 80); a header cut after the first 2 bytes is a + // valid-but-incomplete sequence, not corruption — streaming decode allows it. + const full = Buffer.from("ok 😀", "utf-8"); + const truncated = full.subarray(0, full.length - 2); + expect(truncated.indexOf(0)).toBe(-1); + expect(isProbablyBinaryHeader(truncated)).toBe(false); + }); +}); + +describe("isProbablyBinary / isProbablyBinarySync", () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-binary-")); + + function writeFile(name: string, bytes: Uint8Array | string): string { + const filePath = path.join(tempDir, name); + fs.writeFileSync(filePath, bytes); + return filePath; + } + + it("classifies a binary file from disk (async + sync agree)", async () => { + const filePath = writeFile("font.ttf", Buffer.from([0x00, 0x01, 0x00, 0x00, 0x00, 0x0c])); + expect(await isProbablyBinary(filePath)).toBe(true); + expect(isProbablyBinarySync(filePath)).toBe(true); + }); + + it("classifies a UTF-8 text file from disk as text", async () => { + const filePath = writeFile("notes.md", "# Title\n\nbody text\n"); + expect(await isProbablyBinary(filePath)).toBe(false); + expect(isProbablyBinarySync(filePath)).toBe(false); + }); +}); From 5c382e6b7eef87b15a6c25e6d83510e04cf7bdf1 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 03:06:06 +0200 Subject: [PATCH 19/88] fix(coding-agent): formatted and type-fixed tiny-model download test Reflow biome wrap and replace delete on non-optional isTTY with Reflect.deleteProperty in #3840's text-failure test. --- packages/coding-agent/test/tiny-models-cli.test.ts | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/test/tiny-models-cli.test.ts b/packages/coding-agent/test/tiny-models-cli.test.ts index f3acdd520..26f5ad5b3 100644 --- a/packages/coding-agent/test/tiny-models-cli.test.ts +++ b/packages/coding-agent/test/tiny-models-cli.test.ts @@ -65,12 +65,12 @@ describe("tiny-models download model resolution", () => { }); try { - await expect( - runTinyModelsCommand({ action: "download", model: "lfm2-700m", flags: {} }), - ).rejects.toThrow("One or more tiny title models failed to download"); + await expect(runTinyModelsCommand({ action: "download", model: "lfm2-700m", flags: {} })).rejects.toThrow( + "One or more tiny title models failed to download", + ); } finally { if (isTtyDescriptor) Object.defineProperty(process.stdout, "isTTY", isTtyDescriptor); - else delete (process.stdout as typeof process.stdout & { isTTY?: boolean }).isTTY; + else Reflect.deleteProperty(process.stdout, "isTTY"); } expect(output.join("")).toContain("Failed to download LFM2 700M: runtime install failed."); From 328e22c0791299886aacf0d308b72df0def1ec38 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 01:09:26 +0000 Subject: [PATCH 20/88] fix(prompting): bounded stdout drain on successful probes Even on exit 0 the GPU probe can leave a descendant holding stdout open. Race the EOF wait against a 250ms grace window so the success path cancels the reader instead of blocking until the descendant exits, and unref the prep deadline timer so a one-shot CLI is not held alive by it once all prep work returns. --- .../coding-agent/src/system-prompt.test.ts | 18 +++++++++++++++++- packages/coding-agent/src/system-prompt.ts | 17 ++++++++++++++--- 2 files changed, 31 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index a55f77208..89274c65e 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -14,6 +14,7 @@ async function runProbeScenario(options: { runs: number; sleepSeconds?: number; holdStdoutOpen?: boolean; + descendantHoldsStdout?: boolean; }): Promise { const tempRoot = await fs.mkdtemp(path.join(os.tmpdir(), "omp-gpu-probe-")); try { @@ -25,7 +26,7 @@ async function runProbeScenario(options: { const lspciPath = path.join(binDir, "lspci"); await Bun.write( lspciPath, - '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ "$OMP_GPU_PROBE_HOLD_STDOUT_OPEN" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & wait "$!"; fi\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', + '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ "$OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & exit 0; fi\nif [ "$OMP_GPU_PROBE_HOLD_STDOUT_OPEN" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & wait "$!"; fi\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', ); await fs.chmod(lspciPath, 0o755); @@ -83,6 +84,11 @@ console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt } else { delete env.OMP_GPU_PROBE_HOLD_STDOUT_OPEN; } + if (options.descendantHoldsStdout) { + env.OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT = "true"; + } else { + delete env.OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT; + } const childStartedAt = performance.now(); const child = Bun.spawn([process.execPath, scenarioPath], { stdout: "pipe", stderr: "pipe", env }); @@ -118,4 +124,14 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { // not linger until a descendant holding stdout (sleep 7) exits on its own. expect(result.childElapsedMs).toBeLessThan(6500); }, 15_000); + + it("does not wait on stdout held by a descendant after a successful probe", async () => { + const result = await runProbeScenario({ runs: 1, sleepSeconds: 3, descendantHoldsStdout: true }); + + expect(result.cached).toEqual({ gpu: null }); + // Probe exits 0 immediately but leaves a backgrounded sleep holding the stdout + // pipe. The success path MUST bound the drain wait, not block until sleep exits. + expect(result.elapsedMs).toBeLessThan(2000); + expect(result.childElapsedMs).toBeLessThan(2000); + }, 15_000); }); diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index d409d0e3a..8aab13c2e 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -113,6 +113,8 @@ function parseWmicTable(output: string, header: string): string | null { const SYSTEM_PROMPT_PREP_TIMEOUT_MS = 5000; /** Kept below prep timeout so timed-out probes can still write the null cache before fallback. */ const GPU_PROBE_TIMEOUT_MS = SYSTEM_PROMPT_PREP_TIMEOUT_MS - 500; +/** Drop stdout from a probe descendant that inherited the pipe after the probe exited. */ +const GPU_PROBE_STDOUT_DRAIN_MS = 250; async function runGpuProbe(cmd: string[]): Promise { try { @@ -138,12 +140,17 @@ async function runGpuProbe(cmd: string[]): Promise { stdout += decoder.decode(); })(); const exitCode = await proc.exited; - if (exitCode !== 0) { + // Even on exit 0, a probe wrapper can leave a descendant holding stdout open. + // Bound the EOF wait so getCachedGpu cannot outlive the probe in either path. + const drained = await Promise.race([ + stdoutDone.then(() => "ok" as const).catch(() => "err" as const), + Bun.sleep(GPU_PROBE_STDOUT_DRAIN_MS).then(() => "timeout" as const), + ]); + if (exitCode !== 0 || drained !== "ok") { await stdoutReader.cancel().catch(() => undefined); await stdoutDone.catch(() => undefined); return null; } - await stdoutDone; return stdout; } catch { return null; @@ -531,7 +538,10 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): gpu: undefined as string | undefined, }; - const deadline = Bun.sleep(SYSTEM_PROMPT_PREP_TIMEOUT_MS).then(() => "__timeout__" as const); + const { promise: deadline, resolve: fireDeadline } = Promise.withResolvers<"__timeout__">(); + const deadlineTimer = setTimeout(() => fireDeadline("__timeout__"), SYSTEM_PROMPT_PREP_TIMEOUT_MS); + // Unref so a fast prep does not hold a one-shot CLI alive waiting for this timer. + deadlineTimer.unref(); const timedOut: string[] = []; const failed: Array<{ name: string; error: unknown }> = []; @@ -630,6 +640,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): withDeadline("resolveActiveRepoContext", activeRepoContextPromise, prepDefaults.activeRepoContext), withDeadline("getCachedGpu", gpuPromise, prepDefaults.gpu), ]); + clearTimeout(deadlineTimer); const agentsMdFiles = Array.from(new Set(workspaceTree.agentsMdFiles)).sort().slice(0, AGENTS_MD_LIMIT); if (timedOut.length > 0) { From 531b4d1e8f782c5b12ee7d0f78e55d4d4f061575 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 03:06:27 +0200 Subject: [PATCH 21/88] chore: normalized merged changelog headings --- packages/coding-agent/CHANGELOG.md | 35 ++++++++---------------------- packages/natives/CHANGELOG.md | 6 ++--- packages/snapcompact/CHANGELOG.md | 19 +++++----------- packages/utils/CHANGELOG.md | 2 +- 4 files changed, 17 insertions(+), 45 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d7f5a3d24..a8e76887c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,35 +4,18 @@ ### Changed -- Improved binary file detection to prevent terminal corruption from non-UTF8 content -- Updated file mention summaries to explicitly note skipped binary files - -- Enabled contextual snapcompact shape resolution based on rendered text content +- Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files. +- Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content. ### Fixed -- Fixed snapcompact preflight to use the same font-aware renderability probe as compaction, including prior preserved archive text, so CJK history remains renderable through per-glyph Silver fallback across repeated compactions. -### Fixed - -- Fixed in-TUI `/resume` materializing the previous session's display context (snapcompact archives + OpenAI Responses replay payloads) before switching files, which could exhaust memory on huge pre-fix sessions. `AgentSession.switchSession` now only snapshots the prior context for same-session reloads, where it is needed for rollback comparison. ([#3846](https://github.com/can1357/oh-my-pi/issues/3846)) -### Fixed - -- Fixed `irc` inbox drains missing messages that arrived while the recipient agent was already running. ([#3834](https://github.com/can1357/oh-my-pi/issues/3834)) -### Fixed - -- Fixed system-prompt GPU detection blocking startup and rerunning failed probes by applying the prep deadline and caching empty results. ([#3835](https://github.com/can1357/oh-my-pi/issues/3835)) -### Fixed - -- Fixed `omp tiny-models download` JSON/text failures to include the worker-side download error instead of collapsing every worker failure to `ok:false`. ([#3839](https://github.com/can1357/oh-my-pi/issues/3839)) -### Fixed - -- Fixed `/extensions` showing MCP servers as `active` when `/mcp list` reported them as `disabled`, and made `/extensions` re-enable work for every supported source. The dashboard now mirrors `/mcp list` by honoring the per-server `enabled: false` flag plus the user-level `disabledServers` denylist and a new `enabledServers` allowlist; the dashboard's MCP toggle writes through the canonical mcp.json — flipping `enabled` on the loaded source file for config-resident servers (including supported non-primary files such as `.omp/.mcp.json`), force-enabling tool-owned sources (such as `opencode.json`) via the user `enabledServers` allowlist without mutating the foreign config, and using the `disabledServers` denylist for purely discovered third-party servers, so `/mcp list`, the MCP runtime, and the dashboard stay in sync ([#3827](https://github.com/can1357/oh-my-pi/issues/3827)). -### Fixed - -- Isolated branch-mode task merges now preserve the agent's own commits (message and author) instead of collapsing every diff into a single AI-summarized commit. `commitToBranch` detects when the subagent moved HEAD past the baseline, transfers clean-baseline commit objects into the parent repo via `git fetch`, rewrites dirty-baseline commits against the captured baseline WIP so user staged/unstaged/untracked changes are filtered out, and `mergeTaskBranches` cherry-picks the inclusive range `baseSha..omp/task/` so each task commit replays with its original message; any uncommitted leftover on top of the agent's last commit lands as one trailing AI-summarized commit ([#3842](https://github.com/can1357/oh-my-pi/issues/3842)). -### Fixed - -- Fixed isolated branch merges rejecting task edits when the parent checkout had unrelated dirty changes in nearby patch context. ([#3841](https://github.com/can1357/oh-my-pi/issues/3841)) +- Fixed an issue where CJK (Chinese, Japanese, Korean) history could become unrenderable during repeated context compactions. +- Fixed a memory exhaustion bug in the TUI when using `/resume` on large previous sessions. +- Fixed an issue where the `irc` inbox missed messages that arrived while the recipient agent was already running. +- Fixed a startup hang caused by system-prompt GPU detection blocking and repeatedly running failed probes. +- Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error. +- Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized. +- Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout. ## [16.2.6] - 2026-06-29 diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index fa3d88637..e2b7abe82 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -4,10 +4,8 @@ ### Added -- Added support for the silver TrueType font to renderSnapcompactPng -- Exposed snapcompactSupportedChars to check font capability for characters - -- Added embedded Silver TrueType rendering to `renderSnapcompactPng` plus `snapcompactSupportedChars(font, chars)`, including per-glyph Silver fallback when bitmap snapcompact fonts lack a renderable glyph; East Asian wide code points render full-width across two grid cells, with a soft gray fringe on black-ink frames so scaled glyphs read anti-aliased rather than bold. +- Added embedded Silver TrueType font rendering support to `renderSnapcompactPng`, featuring automatic per-glyph fallback for missing bitmap characters and anti-aliased scaling for East Asian wide code points. +- Added `snapcompactSupportedChars` to check font capability for specific characters. ## [16.2.5] - 2026-06-28 diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index 71f3c4d81..fa3a517e9 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -4,23 +4,14 @@ ### Added -- Added `silver16-bw` shape with Silver TrueType font support for CJK and non-Latin text -- Added `resolveShapeForText` to support font-aware shape resolution -- Added semantic emoji folding to ASCII labels (e.g., [OK], [WARN], [FAIL]) -- Added support for East Asian wide characters across two grid cells in bitmap shapes - -- Added the `silver16-bw` snapcompact shape backed by the embedded Silver TrueType font for CJK and other non-Latin text. +- Added the `silver16-bw` shape backed by an embedded Silver TrueType font to support CJK and other non-Latin text. +- Added `resolveShapeForText` to support font-aware shape resolution. ### Changed -- Improved normalization to preserve Unicode glyphs supported by the system font or Silver fallback -- Updated `wrap` and pagination logic to account for wide Character footprint in bitmap shapes -- Dropped decorative emoji from rendered output instead of printing fallbacks -- Folded box-drawing and compatibility symbols to their ASCII skeletons -- Updated provider shape geometries to use updated X.org 8x13 font metrics (11px/22px pitches) - -- Improved snapcompact normalization for non-ASCII text: semantic emoji fold to ASCII labels, decorative emoji drop, box drawing and compatibility symbols keep ASCII folds, and Unicode text is preserved when either the selected font or the embedded Silver fallback can render it. -- Bitmap snapcompact shapes now draw missing glyphs through the embedded Silver TrueType fallback per character instead of switching entire snippets or rendering blanks; East Asian wide characters render full-width across two grid cells (TypeScript capacity/pagination and the native renderer share one wide-character cell model). +- Improved text normalization for non-ASCII text: semantic emojis fold to ASCII labels (e.g., `[OK]`, `[WARN]`, `[FAIL]`), decorative emojis are dropped, box-drawing/compatibility symbols fold to ASCII skeletons, and Unicode text is preserved when supported by the selected font or the embedded Silver fallback. +- Updated bitmap shapes to draw missing glyphs per-character using the embedded Silver TrueType fallback instead of rendering blanks or switching entire snippets, with support for East Asian wide characters across two grid cells. +- Updated text wrapping, pagination, and provider shape geometries to account for wide character footprints and updated X.org 8x13 font metrics (11px/22px pitches). ## [16.1.23] - 2026-06-26 diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 9b6dd0d44..0b05df86e 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added utility to detect binary files based on content sniffing +- Added a utility to detect binary files based on content sniffing. ## [16.2.6] - 2026-06-29 From 0ae97347082b93a1bffdccafc98a6b6025675786 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 01:34:11 +0000 Subject: [PATCH 22/88] fix(prompting): kept gpu probe output when descendants delay eof On a probe that exits 0 with valid output but leaves a descendant holding stdout open, the bounded drain previously discarded the captured bytes and cached { gpu: null }. Use the already-captured stdout when the probe itself succeeded; only treat a non-zero/timeout exit as a failure. --- .../coding-agent/src/system-prompt.test.ts | 23 ++++++++++++++++++- packages/coding-agent/src/system-prompt.ts | 8 +++---- 2 files changed, 26 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index 89274c65e..456d1964f 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -15,6 +15,7 @@ async function runProbeScenario(options: { sleepSeconds?: number; holdStdoutOpen?: boolean; descendantHoldsStdout?: boolean; + validOutput?: string; }): Promise { const tempRoot = await fs.mkdtemp(path.join(os.tmpdir(), "omp-gpu-probe-")); try { @@ -26,7 +27,7 @@ async function runProbeScenario(options: { const lspciPath = path.join(binDir, "lspci"); await Bun.write( lspciPath, - '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ "$OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & exit 0; fi\nif [ "$OMP_GPU_PROBE_HOLD_STDOUT_OPEN" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & wait "$!"; fi\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', + '#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ -n "$OMP_GPU_PROBE_VALID_OUTPUT" ]; then printf "%s\\n" "$OMP_GPU_PROBE_VALID_OUTPUT"; fi\nif [ "$OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & exit 0; fi\nif [ "$OMP_GPU_PROBE_HOLD_STDOUT_OPEN" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & wait "$!"; fi\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n', ); await fs.chmod(lspciPath, 0o755); @@ -89,6 +90,11 @@ console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt } else { delete env.OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT; } + if (options.validOutput !== undefined) { + env.OMP_GPU_PROBE_VALID_OUTPUT = options.validOutput; + } else { + delete env.OMP_GPU_PROBE_VALID_OUTPUT; + } const childStartedAt = performance.now(); const child = Bun.spawn([process.execPath, scenarioPath], { stdout: "pipe", stderr: "pipe", env }); @@ -134,4 +140,19 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { expect(result.elapsedMs).toBeLessThan(2000); expect(result.childElapsedMs).toBeLessThan(2000); }, 15_000); + + it("keeps probe output captured before a descendant delays EOF", async () => { + const result = await runProbeScenario({ + runs: 1, + sleepSeconds: 3, + descendantHoldsStdout: true, + validOutput: "00:02.0 VGA compatible controller: NVIDIA TestGPU", + }); + + // Probe exited 0 with valid output before bg sleep held stdout open. + // Captured stdout MUST be cached, not discarded as if the probe failed. + expect(result.cached).toEqual({ gpu: "02.0 VGA compatible controller: NVIDIA TestGPU" }); + expect(result.elapsedMs).toBeLessThan(2000); + expect(result.childElapsedMs).toBeLessThan(2000); + }, 15_000); }); diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 8aab13c2e..3c410b396 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -141,17 +141,17 @@ async function runGpuProbe(cmd: string[]): Promise { })(); const exitCode = await proc.exited; // Even on exit 0, a probe wrapper can leave a descendant holding stdout open. - // Bound the EOF wait so getCachedGpu cannot outlive the probe in either path. + // Bound the EOF wait so getCachedGpu cannot outlive the probe in either path; + // keep whatever bytes the reader already captured before cancelling. const drained = await Promise.race([ stdoutDone.then(() => "ok" as const).catch(() => "err" as const), Bun.sleep(GPU_PROBE_STDOUT_DRAIN_MS).then(() => "timeout" as const), ]); - if (exitCode !== 0 || drained !== "ok") { + if (drained !== "ok") { await stdoutReader.cancel().catch(() => undefined); await stdoutDone.catch(() => undefined); - return null; } - return stdout; + return exitCode === 0 ? stdout : null; } catch { return null; } From d4be774eb2310523703126f6d1daeeb5be0ea49f Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 03:35:30 +0200 Subject: [PATCH 23/88] chore: update stale tests --- packages/coding-agent/test/agent-session-handoff.test.ts | 5 +++-- .../test/agent-session-snapcompact-auto-fallback.test.ts | 6 ++++-- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index 52251ec70..d8e82aa54 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -17,6 +17,7 @@ import { TempDir } from "@oh-my-pi/pi-utils"; import * as snapcompact from "@oh-my-pi/snapcompact"; const HANDOFF_SECRET = "HANDOFF_SECRET_TOKEN_12345"; +const UNRENDERABLE_SNAPCOMPACT_TEXT = "\uE000\uE001\uE002\uE003\uE004\uE005\uE006\uE007\uE008\uE009"; describe("AgentSession handoff", () => { // Immutable across the whole file: the model registry's synchronous bundled-model @@ -451,7 +452,7 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: "中文内容".repeat(100) }], timestamp: 1 }, + { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, ], turnPrefixMessages: [], recentMessages: [], @@ -478,7 +479,7 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: "中文内容".repeat(100) }], timestamp: 1 }, + { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, ], turnPrefixMessages: [], recentMessages: [], diff --git a/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts b/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts index 3a3a4ef38..0a9af2702 100644 --- a/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts @@ -11,6 +11,8 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { TempDir } from "@oh-my-pi/pi-utils"; +const UNRENDERABLE_SNAPCOMPACT_TEXT = "\uE000\uE001\uE002\uE003\uE004\uE005\uE006\uE007\uE008\uE009"; + interface Harness { session: AgentSession; sessionManager: SessionManager; @@ -42,7 +44,7 @@ async function createHarness(tempDir: TempDir, authStorage: AuthStorage, options const settings = Settings.isolated({ "compaction.strategy": "snapcompact", // Force a 1-token recent window so the post-turn cut always splits off the - // last turn and summarizes the seeded (CJK) history. With the default + // last turn and summarizes the seeded unrenderable history. With the default // 20k window the cut keeps both tiny messages, leaving nothing for // snapcompact's renderability preflight to scan. "compaction.keepRecentTokens": 1, @@ -153,7 +155,7 @@ describe("AgentSession auto-snapcompact local-blocker fallback", () => { seedMessages: [ { role: "user", - content: "你好,请帮我审查这段代码。它的逻辑似乎有问题,我无法理解为何返回空结果。", + content: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(10), timestamp: Date.now(), }, ], From 1db1e9e200b8e610811d562c0bd0d354308a4334 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 02:12:51 +0000 Subject: [PATCH 24/88] fix(catalog): omitted disabled thinking for kimi code Kimi K2.7 Code rejects disabled thinking on native Kimi endpoints, so route caller disable requests through the omit mode and let the model default to required thinking. Added regression coverage for the title-generator-style Kimi Code request, Moonshot K2.7 Code variants, and K2.6's still-supported disabled-thinking path. Fixes #3852 --- .../__tests__/kimi-code-thinking.test.ts | 52 +++ packages/catalog/CHANGELOG.md | 4 + packages/catalog/src/compat/openai.ts | 12 +- packages/catalog/src/models.json | 438 ++++++++++-------- .../src/provider-models/openai-compat.ts | 2 +- 5 files changed, 303 insertions(+), 205 deletions(-) create mode 100644 packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts diff --git a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts new file mode 100644 index 000000000..0b3c08de4 --- /dev/null +++ b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "bun:test"; +import { getBundledModel } from "@oh-my-pi/pi-catalog"; +import { + applyChatCompletionsCompatPolicy, + type OpenAICompletionsParams, + resolveOpenAICompatPolicy, +} from "../openai-shared"; + +const BASE_CHAT_COMPLETIONS_PARAMS: OpenAICompletionsParams = { messages: [], model: "unused", stream: true }; + +describe("Kimi K2.7 Code thinking policy", () => { + it("omits disabled thinking for title-generator-style Kimi Code requests", () => { + const model = getBundledModel("kimi-code", "kimi-for-coding"); + const policy = resolveOpenAICompatPolicy(model, { + endpoint: "chat-completions", + disableReasoning: true, + toolChoice: { type: "tool", name: "set_title" }, + }); + const params = { ...BASE_CHAT_COMPLETIONS_PARAMS }; + + applyChatCompletionsCompatPolicy(params, policy); + + expect("thinking" in params).toBe(false); + }); + + it("omits disabled thinking for native Moonshot Kimi K2.7 Code variants", () => { + for (const modelId of ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"]) { + const model = getBundledModel("moonshot", modelId); + const policy = resolveOpenAICompatPolicy(model, { + endpoint: "chat-completions", + disableReasoning: true, + }); + const params = { ...BASE_CHAT_COMPLETIONS_PARAMS }; + applyChatCompletionsCompatPolicy(params, policy); + + expect("thinking" in params).toBe(false); + } + }); + + it("keeps explicit disabled thinking for Kimi K2.6", () => { + const model = getBundledModel("moonshot", "kimi-k2.6"); + const policy = resolveOpenAICompatPolicy(model, { + endpoint: "chat-completions", + disableReasoning: true, + }); + const params = { ...BASE_CHAT_COMPLETIONS_PARAMS }; + + applyChatCompletionsCompatPolicy(params, policy); + + expect(params.thinking).toEqual({ type: "disabled" }); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 09b0580c6..dc60dc92a 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Kimi K2.7 Code compatibility to avoid sending disabled thinking to native Kimi endpoints that require thinking mode. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 26ae8b357..42ab2f305 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -39,6 +39,13 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; /** Kimi K2.6 can spend several minutes reasoning before the first visible token. */ const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; +/** Native Kimi K2.7 Code requires thinking; sending `thinking: { type: "disabled" }` 400s. */ +const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i; + +function requiresKimiK27CodeEnabledThinking(spec: ModelSpec<"openai-completions">): boolean { + if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true; + return spec.provider === "kimi-code" && spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); +} /** Xiaomi MiMo Pro on api.xiaomimimo.com can stall ~2min before the first event (issue #1770). */ const XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS = 300_000; /** Alibaba Coding Plan (coding-intl.dashscope) qwen models idle before the first event (issue #1770). */ @@ -231,6 +238,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const isKimiModel = isKimiModelId(spec.id); const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative"); const isMoonshotKimi = isKimiModel && isMoonshotNative; + const requiresEnabledThinking = requiresKimiK27CodeEnabledThinking(spec); const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id); const isAnthropicModel = modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id); @@ -509,7 +517,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv applyCompatOverrides(compat, spec.compat); if (spec.compat?.reasoningDisableMode === undefined) { - compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat); + compat.reasoningDisableMode = requiresEnabledThinking + ? "omit" + : resolveReasoningDisableMode(compat.thinkingFormat); } if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 492404b0c..877830895 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -2476,7 +2476,7 @@ "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -2487,24 +2487,7 @@ "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 16384, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } + "maxTokens": 16384 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -10620,6 +10603,35 @@ ] } }, + "xai.grok-4.3": { + "id": "xai.grok-4.3", + "name": "Grok 4.3", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 2.5, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "zai.glm-4.7": { "id": "zai.glm-4.7", "name": "GLM-4.7", @@ -13672,7 +13684,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -13683,17 +13695,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 128000 }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", @@ -13701,7 +13703,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -13712,17 +13714,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 128000 }, "meta-llama/Llama-4-Scout-17B-16E-Instruct": { "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct", @@ -13730,7 +13722,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" @@ -13742,17 +13734,7 @@ "cacheWrite": 0 }, "contextWindow": 64000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 64000 }, "microsoft/Phi-4-mini-instruct": { "id": "microsoft/Phi-4-mini-instruct", @@ -13760,7 +13742,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -13771,17 +13753,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 128000 }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", @@ -13789,7 +13761,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -13800,7 +13772,16 @@ "cacheWrite": 0 }, "contextWindow": 196608, - "maxTokens": 196608 + "maxTokens": 196608, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", @@ -13898,7 +13879,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -13909,7 +13890,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144 + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", @@ -21734,7 +21725,7 @@ "cost": { "input": 0.075, "output": 0.3, - "cacheRead": 0.037, + "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, @@ -22420,13 +22411,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.25, + "output": 0.69, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -25048,7 +25039,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -25059,24 +25050,7 @@ "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } + "maxTokens": 65536 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -30816,6 +30790,7 @@ "supportsStrictMode": false, "toolStrictMode": "mixed", "stripDeepseekSpecialTokens": false, + "streamMarkupHealingPattern": "thinking", "reasoningDeltasMayBeCumulative": false, "emptyLengthFinishIsContextError": false, "usesOpenAIToolCallIdLimit": false, @@ -54444,6 +54419,36 @@ "requiresEffort": true } }, + "minimaxai/minimax-m3": { + "id": "minimaxai/minimax-m3", + "name": "MiniMax-M3", + "api": "openai-completions", + "provider": "nvidia", + "baseUrl": "https://integrate.api.nvidia.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "mistralai/codestral-22b-instruct-v0.1": { "id": "mistralai/codestral-22b-instruct-v0.1", "name": "Codestral 22b Instruct V0.1", @@ -62195,9 +62200,9 @@ "image" ], "cost": { - "input": 0.66, - "output": 3.41, - "cacheRead": 0.144, + "input": 0.55, + "output": 3.1999999999999997, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 262144, @@ -63522,7 +63527,7 @@ "api": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", "provider": "openrouter", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -63533,22 +63538,7 @@ "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 16384, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } - } + "maxTokens": 16384 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -65751,7 +65741,7 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 32768 + "maxTokens": 100352 }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", @@ -65770,7 +65760,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144 + "maxTokens": 100352 }, "moonshotai/kimi-k2-0905:exacto": { "id": "moonshotai/kimi-k2-0905:exacto", @@ -65861,9 +65851,9 @@ "image" ], "cost": { - "input": 0.66, - "output": 3.41, - "cacheRead": 0.144, + "input": 0.55, + "output": 3.1999999999999997, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 262144, @@ -69347,8 +69337,8 @@ "image" ], "cost": { - "input": 0.28850000000000003, - "output": 2.65, + "input": 0.2596, + "output": 2.3850000000000002, "cacheRead": 0.15, "cacheWrite": 0 }, @@ -69758,13 +69748,13 @@ "text" ], "cost": { - "input": 0.09, + "input": 0.09999999999999999, "output": 0.3, "cacheRead": 0.02, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -70846,8 +70836,8 @@ "text" ], "cost": { - "input": 0.98, - "output": 3.08, + "input": 0.975, + "output": 4.300000000000001, "cacheRead": 0.182, "cacheWrite": 0 }, @@ -70874,7 +70864,7 @@ "text" ], "cost": { - "input": 0.95, + "input": 0.94, "output": 3, "cacheRead": 0.18, "cacheWrite": 0 @@ -71096,7 +71086,7 @@ "synthetic": { "hf:MiniMaxAI/MiniMax-M3": { "id": "hf:MiniMaxAI/MiniMax-M3", - "name": "MiniMaxAI/MiniMax-M3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71111,7 +71101,7 @@ "cacheRead": 0.6, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 524288, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -71126,7 +71116,7 @@ }, "hf:moonshotai/Kimi-K2.6": { "id": "hf:moonshotai/Kimi-K2.6", - "name": "moonshotai/Kimi-K2.6", + "name": "Kimi K2.6", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71156,7 +71146,7 @@ }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", - "name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", + "name": "Nemotron 3 Super 120B A12B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71185,7 +71175,7 @@ }, "hf:openai/gpt-oss-120b": { "id": "hf:openai/gpt-oss-120b", - "name": "openai/gpt-oss-120b", + "name": "GPT OSS 120B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71196,7 +71186,7 @@ "cost": { "input": 0.1, "output": 0.1, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 131072, @@ -71212,7 +71202,7 @@ }, "hf:Qwen/Qwen3.5-397B-A17B": { "id": "hf:Qwen/Qwen3.5-397B-A17B", - "name": "Qwen/Qwen3.5-397B-A17B", + "name": "Qwen3.5 397B-A17B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71223,7 +71213,7 @@ ], "cost": { "input": 0.6, - "output": 3, + "output": 3.6, "cacheRead": 0.6, "cacheWrite": 0 }, @@ -71241,26 +71231,36 @@ }, "hf:Qwen/Qwen3.6-27B": { "id": "hf:Qwen/Qwen3.6-27B", - "name": "Qwen/Qwen3.6-27B", + "name": "Qwen3.6 27B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.45, + "output": 3.6, + "cacheRead": 0.45, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 8192 + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } }, "hf:zai-org/GLM-4.7": { "id": "hf:zai-org/GLM-4.7", - "name": "zai-org/GLM-4.7", + "name": "GLM-4.7", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71269,13 +71269,13 @@ "text" ], "cost": { - "input": 0.55, + "input": 0.45, "output": 2.19, - "cacheRead": 0, + "cacheRead": 0.45, "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 64000, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -71289,7 +71289,7 @@ }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", - "name": "zai-org/GLM-4.7-Flash", + "name": "GLM-4.7-Flash", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71298,9 +71298,9 @@ "text" ], "cost": { - "input": 0.06, - "output": 0.4, - "cacheRead": 0.06, + "input": 0.1, + "output": 0.5, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 196608, @@ -71318,7 +71318,7 @@ }, "hf:zai-org/GLM-5.1": { "id": "hf:zai-org/GLM-5.1", - "name": "zai-org/GLM-5.1", + "name": "GLM-5.1", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71347,22 +71347,32 @@ }, "hf:zai-org/GLM-5.2": { "id": "hf:zai-org/GLM-5.2", - "name": "zai-org/GLM-5.2", + "name": "GLM-5.2", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.4, + "output": 4.4, + "cacheRead": 1.4, "cacheWrite": 0 }, "contextWindow": 524288, - "maxTokens": 8192 + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "syn:large:text": { "id": "syn:large:text", @@ -71682,7 +71692,7 @@ "api": "openai-completions", "provider": "together", "baseUrl": "https://api.together.xyz/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" @@ -71694,17 +71704,7 @@ "cacheWrite": 0.18 }, "contextWindow": 10000000, - "maxTokens": 32768, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 32768 }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", @@ -72381,6 +72381,38 @@ "escapeBuiltinToolNames": true } }, + "umans-glm-5.2-nvfp4": { + "id": "umans-glm-5.2-nvfp4", + "name": "Umans GLM 5.2 NVFP4 (experimental, short test from Jun 29)", + "api": "anthropic-messages", + "provider": "umans", + "baseUrl": "https://api.code.umans.ai", + "reasoning": true, + "thinking": { + "mode": "anthropic-budget-effort", + "efforts": [ + "high", + "xhigh" + ], + "effortMap": { + "xhigh": "max" + } + }, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 405504, + "maxTokens": 131071, + "compat": { + "escapeBuiltinToolNames": true + } + }, "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", "name": "Umans Kimi K2.7 Code", @@ -80633,7 +80665,7 @@ }, "zai/glm-4.5": { "id": "zai/glm-4.5", - "name": "GLM-4.5", + "name": "GLM 4.5", "api": "anthropic-messages", "baseUrl": "https://ai-gateway.vercel.sh", "provider": "vercel-ai-gateway", @@ -82062,6 +82094,14 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false + }, "thinking": { "mode": "effort", "efforts": [ @@ -82074,14 +82114,6 @@ "effortMap": { "minimal": "low" } - }, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false } }, "grok-4.3": { @@ -82103,6 +82135,14 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false + }, "thinking": { "mode": "effort", "efforts": [ @@ -82115,14 +82155,6 @@ "effortMap": { "minimal": "low" } - }, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false } }, "grok-build": { @@ -82174,12 +82206,12 @@ "contextWindow": 256000, "maxTokens": 256000, "compat": { - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": true, "reasoningEffortMap": { "minimal": "low" }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": true, "supportsReasoningEffort": false } }, @@ -82223,9 +82255,9 @@ "text" ], "cost": { - "input": 0.1, - "output": 0.3, - "cacheRead": 0.01, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 262144, @@ -82259,9 +82291,9 @@ "image" ], "cost": { - "input": 0.4, - "output": 2, - "cacheRead": 0.08, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 262144, @@ -82294,9 +82326,9 @@ "text" ], "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.2, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.0036, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -82330,9 +82362,9 @@ "image" ], "cost": { - "input": 0.4, - "output": 2, - "cacheRead": 0.08, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -82365,9 +82397,9 @@ "text" ], "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.2, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.0036, "cacheWrite": 0 }, "contextWindow": 1048576, diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 12776ea48..ea448e488 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1294,7 +1294,7 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | export const KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS = 32_768; export function isKimiK27CodeModelId(modelId: string): boolean { - return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code$/i.test(modelId); + return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i.test(modelId); } export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number): number; From d20e6c08294ea0b7202cbeb1774f29f7f4f4a17d Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 04:10:36 +0200 Subject: [PATCH 25/88] feat: migrated service tier settings to a per-model-family architecture - Migrated global service tier settings to a per-model-family architecture (OpenAI, Anthropic, Google). - Implemented `ServiceTierByFamily` mapping to allow independent configuration and resolution per provider. - Added automatic migration logic for legacy service tier and fast-mode application settings. - Updated telemetry, session management, and task execution to support provider-specific tier resolution. --- docs/session-switching-and-recent-listing.md | 2 +- docs/session.md | 4 +- docs/settings.md | 6 +- packages/agent/src/compaction/entries.ts | 4 +- packages/agent/src/telemetry.ts | 4 +- packages/ai/CHANGELOG.md | 17 ++ packages/ai/src/providers/anthropic.ts | 10 +- packages/ai/src/providers/google-shared.ts | 13 + packages/ai/src/providers/google-types.ts | 7 + packages/ai/src/providers/google-vertex.ts | 32 ++- .../ai/src/providers/openai-chat-server.ts | 4 +- packages/ai/src/providers/openai-shared.ts | 16 +- packages/ai/src/stream.ts | 5 + packages/ai/src/types.ts | 199 ++++++++++---- packages/ai/test/anthropic-fast-mode.test.ts | 16 -- packages/ai/test/google-service-tier.test.ts | 105 ++++++++ .../service-tier-premium-requests.test.ts | 251 +++++++++--------- packages/coding-agent/CHANGELOG.md | 5 + packages/coding-agent/src/cli/bench-cli.ts | 31 ++- packages/coding-agent/src/commands/bench.ts | 6 +- .../coding-agent/src/config/service-tier.ts | 149 ++++++----- .../src/config/settings-schema.ts | 84 +++--- packages/coding-agent/src/config/settings.ts | 47 ++++ .../coding-agent/src/eval/agent-bridge.ts | 6 +- packages/coding-agent/src/main.ts | 2 +- packages/coding-agent/src/sdk.ts | 23 +- .../coding-agent/src/session/agent-session.ts | 168 +++++++----- .../src/session/session-context.ts | 8 +- .../src/session/session-entries.ts | 4 +- .../src/session/session-manager.ts | 11 +- .../src/slash-commands/builtin-registry.ts | 12 +- packages/coding-agent/src/task/executor.ts | 39 +-- packages/coding-agent/src/task/index.ts | 12 +- packages/coding-agent/src/tools/index.ts | 6 +- .../test/agent-session-handoff.test.ts | 12 +- .../test/agent-session-mcp-discovery.test.ts | 10 +- .../test/bench-auth-fallback.test.ts | 17 +- .../coding-agent/test/fast-mode-scope.test.ts | 75 ++---- .../test/sdk-mcp-discovery.test.ts | 12 +- .../test/service-tier-migration.test.ts | 82 ++++++ packages/stats/CHANGELOG.md | 4 + packages/stats/src/parser.ts | 24 +- packages/stats/src/types.ts | 4 +- 43 files changed, 1015 insertions(+), 533 deletions(-) create mode 100644 packages/ai/test/google-service-tier.test.ts create mode 100644 packages/coding-agent/test/service-tier-migration.test.ts diff --git a/docs/session-switching-and-recent-listing.md b/docs/session-switching-and-recent-listing.md index 24987c6c7..51dc768fd 100644 --- a/docs/session-switching-and-recent-listing.md +++ b/docs/session-switching-and-recent-listing.md @@ -179,7 +179,7 @@ Lifecycle/state transition: 15. restore model via `getRestorableSessionModels(sessionContext.models, lastModelChangeRole)` — tries the recorded models in fallback order and uses the first one present in the model registry 16. restore thinking level and service tier: - thinking uses persisted `thinking_level_change`, otherwise the configured default clamped to model capability - - service tier uses persisted `service_tier_change`, otherwise the configured `serviceTier` setting (`"none"` becomes unset) + - service tier uses persisted `service_tier_change`, otherwise the configured per-family `tier.openai`/`tier.anthropic`/`tier.google` settings (`"none"` becomes unset) 17. reconnect agent listeners, run the registered session-switch reconciler if any (interactive mode re-enters persisted modes; errors logged, not fatal), and return `true` ## UI state rebuild after interactive switch diff --git a/docs/session.md b/docs/session.md index 1103919fa..84182abe3 100644 --- a/docs/session.md +++ b/docs/session.md @@ -177,11 +177,11 @@ Stores an `AgentMessage` directly. "id": "c1d2e3f4", "parentId": "b1c2d3e4", "timestamp": "2026-02-16T10:21:45.000Z", - "serviceTier": "flex" + "serviceTier": { "openai": "priority", "google": "flex" } } ``` -`serviceTier` can also be `null`. +`serviceTier` is a per-family map keyed by `openai`/`anthropic`/`google` (each value `auto`/`default`/`flex`/`scale`/`priority`), or `null` when no tier is active. Legacy entries that stored a single string (`"flex"`, `"openai-only"`, `"claude-only"`, …) are normalized to this map on read. ### `thinking_level_change` diff --git a/docs/settings.md b/docs/settings.md index 71c36134b..fdf61ce83 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -366,7 +366,11 @@ A value of `-1` means "use the provider/model default" — `omp` does not send t | `minP` | number | `-1` | Minimum-probability cutoff. | | `presencePenalty` | number | `-1` | Presence penalty. | | `repetitionPenalty` | number | `-1` | Repetition penalty. | -| `serviceTier` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`, `openai-only`, `claude-only`. | +| `tier.openai` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`. Sent as `service_tier` for OpenAI / OpenAI-Codex and OpenAI-family OpenRouter models. | +| `tier.anthropic` | enum | `none` | `none`, `priority`. `priority` realizes fast mode on supported direct Claude models (ignored on Bedrock/Vertex and via OpenRouter). | +| `tier.google` | enum | `none` | `none`, `flex`, `priority`. Gemini API sends it in the body; Vertex sends `priority` via header (`flex` is a no-op on Vertex). | +| `tier.subagent` | enum | `inherit` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the spawned model's family; `inherit` tracks the main agent. | +| `tier.advisor` | enum | `none` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the advisor model's family. | | `personality` | enum | `default` | `default`, `friendly`, `pragmatic`, `none`. | ### Retry and fallback diff --git a/packages/agent/src/compaction/entries.ts b/packages/agent/src/compaction/entries.ts index 70831433f..f06299990 100644 --- a/packages/agent/src/compaction/entries.ts +++ b/packages/agent/src/compaction/entries.ts @@ -1,4 +1,4 @@ -import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai"; +import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai"; import type { AgentMessage } from "../types"; export interface SessionEntryBase { @@ -28,7 +28,7 @@ export interface ModelChangeEntry extends SessionEntryBase { export interface ServiceTierChangeEntry extends SessionEntryBase { type: "service_tier_change"; - serviceTier: ServiceTier | null; + serviceTier: ServiceTierByFamily | null; } export interface CompactionEntry extends SessionEntryBase { diff --git a/packages/agent/src/telemetry.ts b/packages/agent/src/telemetry.ts index d2a855ba7..e6314a325 100644 --- a/packages/agent/src/telemetry.ts +++ b/packages/agent/src/telemetry.ts @@ -30,7 +30,6 @@ import { completeSimple, type Message, type Model, - resolveServiceTier, type ServiceTier, type SimpleStreamOptions, type StopReason, @@ -752,8 +751,7 @@ function buildChatRequestAttributes(stepNumber: number, request: ChatRequestSnap attrs[GenAIAttr.RequestStopSequences] = [...request.stopSequences]; } if (request.serviceTier && shouldSendServiceTier(request.serviceTier, provider)) { - const resolved = resolveServiceTier(request.serviceTier, provider); - if (resolved) attrs[OpenAIAttr.RequestServiceTier] = resolved; + attrs[OpenAIAttr.RequestServiceTier] = request.serviceTier; } if (request.reasoningEffort) attrs[PiGenAIAttr.RequestReasoningEffort] = request.reasoningEffort; const toolChoice = serializeToolChoice(request.toolChoice); diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3a4f8c78d..7ef1fbf50 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,23 @@ ## [Unreleased] +### Added + +- Added service tier support for Google Gemini and Vertex AI +- Introduced `ServiceTierByFamily` to allow model-specific service tier configurations + +### Changed + +- Updated service tier logic to avoid global scopes in favor of per-provider configurations +- Refactored priority request billing to better align with specific provider capabilities +- Updated internal `coerceServiceTierByFamily` helper to facilitate migration from legacy settings + +### Fixed + +- Fixed safety setting application for Google Vertex AI models +- Ensured Gemini service tier is correctly passed through to the API +- Corrected priority request accounting for supported providers + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 442cfd59e..c89aaaa29 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -43,7 +43,6 @@ import type { ToolResultMessage, Usage, } from "../types"; -import { resolveServiceTier } from "../types"; import { isRecord, normalizeSystemPrompts, normalizeToolCallId, resolveCacheRetention } from "../utils"; import { createAbortSourceTracker } from "../utils/abort"; import { @@ -1595,7 +1594,7 @@ const streamAnthropicOnce = ( isOAuthToken = false; } else { const extraBetas = normalizeExtraBetas(options?.betas); - const wantsAnthropicPriority = resolveServiceTier(options?.serviceTier, model.provider) === "priority"; + const wantsAnthropicPriority = model.provider === "anthropic" && options?.serviceTier === "priority"; // Skip the fast-mode beta when this session already learned the // endpoint+model rejects fast mode; `speed` is dropped from the params // too (dropFastMode), so the request stays a faithful non-fast request. @@ -2191,7 +2190,8 @@ const streamAnthropicOnce = ( } if ( !dropFastMode && - resolveServiceTier(options?.serviceTier, model.provider) === "priority" && + model.provider === "anthropic" && + options?.serviceTier === "priority" && firstTokenTime === undefined && AIError.isFastModeUnsupported(streamFailure) ) { @@ -2259,7 +2259,7 @@ const streamAnthropicOnce = ( } output.duration = performance.now() - startTime; if (firstTokenTime) output.ttft = firstTokenTime - startTime; - if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") { + if (dropFastMode && model.provider === "anthropic" && options?.serviceTier === "priority") { output.disabledFeatures = [...(output.disabledFeatures ?? []), "priority"]; } stream.push({ type: "done", reason: output.stopReason, message: output }); @@ -2996,7 +2996,7 @@ function buildParams( seqs.length > ANTHROPIC_STOP_SEQUENCES_MAX ? seqs.slice(0, ANTHROPIC_STOP_SEQUENCES_MAX) : seqs; } - if (resolveServiceTier(options?.serviceTier, model.provider) === "priority") { + if (model.provider === "anthropic" && options?.serviceTier === "priority") { params.speed = "fast"; } diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index de75ec704..f2fe100d8 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -14,6 +14,7 @@ import type { FetchImpl, ImageContent, Model, + ServiceTier, StopReason, StreamOptions, TextContent, @@ -21,6 +22,7 @@ import type { Tool, ToolCall, } from "../types"; +import { shouldSendServiceTier } from "../types"; import { normalizeSystemPrompts } from "../utils"; import { AssistantMessageEventStream } from "../utils/event-stream"; import type { RawHttpRequestDump } from "../utils/http-inspector"; @@ -73,6 +75,8 @@ export interface GoogleSharedStreamOptions extends StreamOptions { budgetTokens?: number; level?: GoogleThinkingLevel; }; + /** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */ + serviceTier?: ServiceTier; } /** @@ -791,6 +795,14 @@ export function buildGoogleGenerateContentParams 0 && { tools: convertTools(context.tools, model) }), }; + // Gemini API (google-generative-ai) reads the tier from the request body; + // Vertex AI ignores a body field and requires the + // `X-Vertex-AI-LLM-Shared-Request-Type` header instead (added in + // streamGoogleVertex), so only emit the body field for the direct API. + if (model.provider === "google" && shouldSendServiceTier(options.serviceTier, model.provider)) { + config.serviceTier = options.serviceTier; + } + if (context.tools && context.tools.length > 0 && options.toolChoice) { const choice = options.toolChoice; if (typeof choice === "string") { @@ -1007,6 +1019,7 @@ function paramsToWireBody(params: GenerateContentParameters): Record = {}; if (config.temperature !== undefined) gen.temperature = config.temperature; diff --git a/packages/ai/src/providers/google-types.ts b/packages/ai/src/providers/google-types.ts index 086448dea..a5bb299ec 100644 --- a/packages/ai/src/providers/google-types.ts +++ b/packages/ai/src/providers/google-types.ts @@ -10,6 +10,8 @@ * - The Cloud Code Assist endpoint used by `google-gemini-cli.ts` */ +import type { ServiceTier } from "../types"; + /** Mirror of `@google/genai`'s `FinishReason` string enum. */ export type FinishReason = | "FINISH_REASON_UNSPECIFIED" @@ -131,6 +133,11 @@ export interface GenerateContentConfig { safetySettings?: Array>; cachedContent?: string; thinkingConfig?: ThinkingConfig; + /** + * Gemini/Vertex serving tier. Serialized to the request body root as + * `serviceTier` (camelCase) by the transformer in `google-shared.ts`. + */ + serviceTier?: ServiceTier; abortSignal?: AbortSignal; } diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index 5400da7bd..996e763c3 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -30,17 +30,47 @@ export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( prepare: async (): Promise => { const apiKey = resolveApiKey(options); const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + params.config ||= {}; + if (!params.config.safetySettings) { + params.config.safetySettings = [ + { + category: "HARM_CATEGORY_HATE_SPEECH", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_DANGEROUS_CONTENT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_HARASSMENT", + threshold: "OFF", + }, + ]; + } const baseHeaders: Record = { ...(model.headers ?? {}), ...(options?.headers ?? {}), }; + // Vertex AI ignores a `serviceTier` request-body field (unlike the direct + // Gemini API); priority must travel as a request header. Only `priority` + // has a documented Vertex request control — `flex` has none, so it's a no-op. + if (options?.serviceTier === "priority") { + baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; + } if (apiKey) { const url = `https://aiplatform.googleapis.com/${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { params, url, - headers: { ...baseHeaders, "x-goog-api-key": apiKey }, + headers: { + ...baseHeaders, + "x-goog-api-key": apiKey, + }, fetch: options?.fetch, }; } diff --git a/packages/ai/src/providers/openai-chat-server.ts b/packages/ai/src/providers/openai-chat-server.ts index da2d02627..441138ea9 100644 --- a/packages/ai/src/providers/openai-chat-server.ts +++ b/packages/ai/src/providers/openai-chat-server.ts @@ -13,7 +13,7 @@ import type { Context, ImageContent, Message, - ResolvedServiceTier, + ServiceTier, StopReason, TextContent, Tool, @@ -38,7 +38,7 @@ function isReasoningEffort(value: unknown): value is ReasoningEffort { return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh"; } -function isServiceTier(value: unknown): value is ResolvedServiceTier { +function isServiceTier(value: unknown): value is ServiceTier { return value === "auto" || value === "default" || value === "flex" || value === "scale" || value === "priority"; } diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index a024d79b5..b6cd43daa 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -39,8 +39,6 @@ import { type Model, OPENAI_MAX_OUTPUT_TOKENS, type Provider, - type ResolvedServiceTier, - resolveServiceTier, type ServiceTier, type StopReason, type StreamOptions, @@ -269,14 +267,13 @@ export function resolveOpenAIRequestSetup( } export function applyOpenAIServiceTier( - params: { service_tier?: ResolvedServiceTier | "auto" | "default" | null | undefined }, + params: { service_tier?: ServiceTier | null | undefined }, serviceTier: ServiceTier | null | undefined, provider: Provider | undefined, ): void { if (!shouldSendServiceTier(serviceTier, provider)) return; - const resolved = resolveServiceTier(serviceTier, provider); - if (resolved === "flex" || resolved === "scale" || resolved === "priority") { - params.service_tier = resolved; + if (serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority") { + params.service_tier = serviceTier; } } @@ -315,10 +312,7 @@ export function applyOpenAIResponsesServiceTierCost( // The response echo is authoritative when present (OpenAI may downgrade a // requested priority/flex turn to default under load); only fall back to the // requested tier when the response omits the echo entirely. - const served = - typeof responseServiceTier === "string" - ? responseServiceTier - : resolveServiceTier(requestServiceTier, model.provider); + const served = typeof responseServiceTier === "string" ? responseServiceTier : (requestServiceTier ?? undefined); const multiplier = getOpenAIResponsesServiceTierCostMultiplier(served); if (multiplier === 1) return; usage.cost.input *= multiplier; @@ -623,7 +617,7 @@ export type OpenAICompletionsParams = Omit( if (!reasoning || !model.reasoning) { return castApi<"google-generative-ai">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: false }, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); @@ -1578,6 +1579,7 @@ function mapOptionsForApi( if (googleModel.thinking?.mode === "google-level") { return castApi<"google-generative-ai">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: true, level: mapEffortToGoogleThinkingLevel(effort), @@ -1661,6 +1663,7 @@ function mapOptionsForApi( if (!reasoning || !model.reasoning) { return castApi<"google-vertex">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: false }, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); @@ -1673,6 +1676,7 @@ function mapOptionsForApi( if (geminiModel.thinking?.mode === "google-level") { return castApi<"google-vertex">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: true, level: mapEffortToGoogleThinkingLevel(effort), @@ -1683,6 +1687,7 @@ function mapOptionsForApi( return castApi<"google-vertex">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: true, budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets), diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index f19e736c5..bc1bb0f72 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -105,84 +105,181 @@ export type ToolChoice = export type CacheRetention = "none" | "short" | "long"; /** - * Service tier hint for processing priority / cost control. + * Service tier hint for processing priority / cost control. These are the + * values providers consume on the wire: * - * The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`, - * `"priority"`) are passed through to providers that understand them - * (OpenAI's `service_tier` field directly; Anthropic translates - * `"priority"` into `speed: "fast"` on supported Opus models). + * - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field + * (`flex`/`scale`/`priority`). + * - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier` + * field (`flex`/`priority`). + * - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for + * the OpenAI- and Google-family upstreams it supports and ignores it + * otherwise. + * - Direct Anthropic: `"priority"` is translated into `speed: "fast"` plus the + * fast-mode beta on supported Opus models. Other tiers are ignored. * - * The scoped values target a specific provider family and behave as the - * unscoped value on the matching provider, or `undefined` everywhere else. - * They let users opt into priority on one family without paying premium - * costs on the other when switching models mid-session. - * - * - `"openai-only"` → `"priority"` on `openai` and `openai-codex`; ignored elsewhere. - * - `"claude-only"` → `"priority"` on direct `anthropic` (not Bedrock/Vertex Claude). + * Per-family scoping is expressed by {@link ServiceTierByFamily}, not by + * scoped sentinel values — see {@link serviceTierFamily}. */ -export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "openai-only" | "claude-only"; +export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority"; -/** Resolved tier — one of the values that providers actually consume on the wire. */ -export type ResolvedServiceTier = Exclude; +/** Provider families that expose an independent service-tier knob. */ +export type ServiceTierFamily = "openai" | "anthropic" | "google"; /** - * Resolves a possibly scoped `ServiceTier` to the effective tier for the - * given provider. Scoped values match their target family and otherwise - * collapse to `undefined`; unscoped values pass through unchanged. + * Per-family service-tier selection. A request consults only the entry for the + * family its model belongs to (see {@link resolveModelServiceTier}), so a user + * can opt one family into priority without affecting the others when switching + * models mid-session. */ -export function resolveServiceTier( - serviceTier: ServiceTier | null | undefined, - provider: Provider | undefined, -): ResolvedServiceTier | undefined { - if (!serviceTier) return undefined; - switch (serviceTier) { - case "openai-only": - return provider === "openai" || provider === "openai-codex" ? "priority" : undefined; - case "claude-only": - return provider === "anthropic" ? "priority" : undefined; - default: - return serviceTier; +export type ServiceTierByFamily = Partial>; + +/** + * Classify a model into the service-tier family whose knob governs it, or + * `undefined` when the model exposes no serving-priority control. + * + * OpenRouter models are classified by id namespace (`anthropic/`, `google/`, + * `openai/`); Claude on Bedrock/Vertex (api `anthropic-messages`) is the + * anthropic family even though its provider is `amazon-bedrock`/`google-vertex`. + */ +export function serviceTierFamily(model: Pick): ServiceTierFamily | undefined { + const provider = model.provider; + if (provider === "openrouter") { + const id = model.id.toLowerCase(); + if (id.startsWith("anthropic/")) return "anthropic"; + if (id.startsWith("google/")) return "google"; + if (id.startsWith("openai/")) return "openai"; + return undefined; } + if (provider === "openai" || provider === "openai-codex") return "openai"; + if (model.api === "anthropic-messages") return "anthropic"; + if (provider === "google" || provider === "google-vertex") return "google"; + return undefined; } /** - * True when the (possibly scoped) tier should be sent on the wire as the - * `service_tier` request field for the given provider. OpenAI / OpenAI-Codex - * accept `flex`/`scale`/`priority`; Fireworks Serverless realizes only its - * Priority serving path (`service_tier: "priority"`) on the OpenAI-compatible - * chat-completions endpoint. Unsupported tiers (`"auto"`, `"default"`), other - * providers, and scope mismatches all return false. + * Reduce a per-family tier map to the single wire tier for `model` — the entry + * for the model's family, or `undefined` when the model has no family. + */ +export function resolveModelServiceTier( + tiers: ServiceTierByFamily | null | undefined, + model: Pick, +): ServiceTier | undefined { + if (!tiers) return undefined; + const family = serviceTierFamily(model); + return family ? tiers[family] : undefined; +} + +/** + * True when the tier should be sent on the wire as the provider's service-tier + * request field. OpenAI / OpenAI-Codex accept `flex`/`scale`/`priority`; Google + * (Gemini API + Vertex) and OpenRouter accept `flex`/`priority`; Fireworks + * Serverless realizes only its Priority serving path. Anthropic is absent — it + * realizes `priority` via `speed: "fast"`, not a service-tier field. */ export function shouldSendServiceTier( serviceTier: ServiceTier | null | undefined, provider: Provider | undefined, ): boolean { - const resolved = resolveServiceTier(serviceTier, provider); - if (provider === "openai" || provider === "openai-codex") { - return resolved === "flex" || resolved === "scale" || resolved === "priority"; + if (!serviceTier) return false; + if (provider === "openai" || provider === "openai-codex" || provider === "openrouter") { + return serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority"; } - if (provider === "fireworks") { - return resolved === "priority"; + if (provider === "google") { + return serviceTier === "flex" || serviceTier === "priority"; + } + // Vertex realizes only priority (via header); flex has no documented control. + if (provider === "google-vertex" || provider === "fireworks") { + return serviceTier === "priority"; } return false; } /** - * Premium-request weight contributed by sending priority to a provider - * that supports it. Mirrors GitHub Copilot's `premiumRequests` accounting - * so the "premium requests" stat aggregates priority traffic across the - * OpenAI family and Anthropic fast-mode realizations. + * True when `priority` will actually be realized on the wire for `model`. + * Direct Anthropic realizes fast mode; OpenAI/Google/Fireworks emit the + * service-tier field; OpenRouter realizes it only for its OpenAI- and + * Google-family upstreams. Bedrock/Vertex Claude and OpenRouter Anthropic + * models do not realize priority and return `false`. + */ +export function realizesPriorityServiceTier( + serviceTier: ServiceTier | null | undefined, + model: Pick, +): boolean { + if (serviceTier !== "priority") return false; + if (model.provider === "anthropic") return true; + if (model.provider === "openrouter") { + const family = serviceTierFamily(model); + return family === "openai" || family === "google"; + } + if (model.api === "anthropic-messages") return false; + return shouldSendServiceTier(serviceTier, model.provider); +} + +/** + * Premium-request weight contributed by a priority request to a provider that + * realizes it and bills extra. Mirrors GitHub Copilot's `premiumRequests` + * accounting so the "premium requests" stat aggregates priority traffic across + * the OpenAI family, direct Anthropic fast mode, and Google priority. * - * Returns 1 per resolved priority request, 0 otherwise. + * Returns 1 only when priority is actually realized on the wire for `model` + * (see {@link realizesPriorityServiceTier}) and the provider bills it as a + * premium request. OpenRouter is excluded — it bills per its own pricing, not + * Copilot-premium semantics — as are Bedrock/Vertex Claude, where priority is + * silently dropped. */ export function getPriorityPremiumRequests( serviceTier: ServiceTier | null | undefined, - provider: Provider | undefined, + model: Pick, ): number { - if (resolveServiceTier(serviceTier, provider) !== "priority") return 0; - // Only providers that realize `priority` on the wire bill the user. - // Everywhere else, the field is silently dropped and nothing is charged. - return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0; + if (!realizesPriorityServiceTier(serviceTier, model)) return 0; + const provider = model.provider; + return provider === "openai" || + provider === "openai-codex" || + provider === "anthropic" || + provider === "google" || + provider === "google-vertex" + ? 1 + : 0; +} + +/** + * Coerce a persisted service-tier value to a {@link ServiceTierByFamily}. Newer + * sessions store the family map directly; legacy sessions stored a single + * scalar — `"priority"` applied everywhere, `"openai-only"`/`"claude-only"` + * scoped to one family, and the remaining values were OpenAI-only semantics. + */ +export function coerceServiceTierByFamily(value: unknown): ServiceTierByFamily | undefined { + if (value === null || value === undefined) return undefined; + if (typeof value === "object") { + const src = value as Record; + const out: ServiceTierByFamily = {}; + for (const family of ["openai", "anthropic", "google"] as const) { + const tier = src[family]; + if (tier === "auto" || tier === "default" || tier === "flex" || tier === "scale" || tier === "priority") { + out[family] = tier; + } + } + return Object.keys(out).length > 0 ? out : undefined; + } + switch (value) { + case "priority": + return { openai: "priority", anthropic: "priority", google: "priority" }; + case "openai-only": + return { openai: "priority" }; + case "claude-only": + return { anthropic: "priority" }; + case "auto": + return { openai: "auto" }; + case "default": + return { openai: "default" }; + case "flex": + return { openai: "flex" }; + case "scale": + return { openai: "scale" }; + default: + return undefined; + } } export interface ProviderSessionState { diff --git a/packages/ai/test/anthropic-fast-mode.test.ts b/packages/ai/test/anthropic-fast-mode.test.ts index b70601c8a..a1d8fa904 100644 --- a/packages/ai/test/anthropic-fast-mode.test.ts +++ b/packages/ai/test/anthropic-fast-mode.test.ts @@ -88,22 +88,6 @@ describe("Anthropic priority service tier → speed='fast'", () => { expect(payload.speed).toBeUndefined(); } }); - - it("sets speed='fast' on direct anthropic when serviceTier='claude-only'", async () => { - const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), { - serviceTier: "claude-only", - })) as { speed?: string }; - expect(payload.speed).toBe("fast"); - }); - - it("omits speed when serviceTier='openai-only' on an anthropic model", async () => { - // Scoped to OpenAI — on this anthropic request, the scope doesn't match, - // so `speed` must not be set on the wire. - const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), { - serviceTier: "openai-only", - })) as Record; - expect(payload.speed).toBeUndefined(); - }); }); describe("clearAnthropicFastModeFallback", () => { diff --git a/packages/ai/test/google-service-tier.test.ts b/packages/ai/test/google-service-tier.test.ts new file mode 100644 index 000000000..15343cf84 --- /dev/null +++ b/packages/ai/test/google-service-tier.test.ts @@ -0,0 +1,105 @@ +import { describe, expect, it } from "bun:test"; +import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; +import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; +import type { AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 1 }] }; + +function sseStop(): Response { + const chunk = { + candidates: [{ content: { parts: [{ text: "ok" }] }, finishReason: "STOP" }], + usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 }, + }; + return new Response(`data: ${JSON.stringify(chunk)}\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +async function drain(stream: AsyncIterable): Promise { + for await (const _ of stream) { + // consume + } +} + +interface Captured { + headers: Headers; + body: Record; +} + +function capturingFetch(): { fetch: FetchImpl; captured: () => Captured } { + let cap: Captured | undefined; + const fetch: FetchImpl = async (_url, init) => { + cap = { + headers: new Headers(init?.headers), + body: JSON.parse(String(init?.body ?? "{}")), + }; + return sseStop(); + }; + return { + fetch, + captured: () => { + if (!cap) throw new Error("fetch was not called"); + return cap; + }, + }; +} + +const geminiModel: Model<"google-generative-ai"> = buildModel({ + id: "gemini-3-flash", + name: "Gemini 3 Flash", + api: "google-generative-ai", + provider: "google", + baseUrl: "https://generativelanguage.googleapis.com/v1beta", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 32_000, +}); + +const vertexModel: Model<"google-vertex"> = buildModel({ + id: "gemini-3-flash", + name: "Gemini 3 Flash (Vertex)", + api: "google-vertex", + provider: "google-vertex", + baseUrl: "", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 32_000, +}); + +describe("Google service tier wire encoding", () => { + it("Gemini API sends the tier in the request body, not a header", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch })); + const { headers, body } = captured(); + expect(body.serviceTier).toBe("priority"); + expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); + }); + + it("Vertex sends priority via header and omits the body tier field", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch })); + const { headers, body } = captured(); + expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBe("priority"); + expect(body.serviceTier).toBeUndefined(); + }); + + it("Vertex omits both header and body for flex (no documented control)", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "flex", fetch })); + const { headers, body } = captured(); + expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); + expect(body.serviceTier).toBeUndefined(); + }); + + it("omits the tier entirely when unset", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch })); + expect(captured().body.serviceTier).toBeUndefined(); + }); +}); diff --git a/packages/ai/test/service-tier-premium-requests.test.ts b/packages/ai/test/service-tier-premium-requests.test.ts index 17f12c1cc..7a217d3be 100644 --- a/packages/ai/test/service-tier-premium-requests.test.ts +++ b/packages/ai/test/service-tier-premium-requests.test.ts @@ -1,138 +1,149 @@ import { describe, expect, it } from "bun:test"; -import { getPriorityPremiumRequests, resolveServiceTier, shouldSendServiceTier } from "@oh-my-pi/pi-ai/types"; +import type { Api } from "@oh-my-pi/pi-ai/types"; +import { + coerceServiceTierByFamily, + getPriorityPremiumRequests, + realizesPriorityServiceTier, + resolveModelServiceTier, + serviceTierFamily, + shouldSendServiceTier, +} from "@oh-my-pi/pi-ai/types"; -describe("getPriorityPremiumRequests", () => { - it("counts priority tier as one premium request on OpenAI", () => { - expect(getPriorityPremiumRequests("priority", "openai")).toBe(1); +const m = (provider: string, api: Api, id: string): { provider: string; api: Api; id: string } => ({ + provider, + api, + id, +}); + +const openai = m("openai", "openai-responses", "gpt-5"); +const codex = m("openai-codex", "openai-codex-responses", "gpt-5.5"); +const anthropic = m("anthropic", "anthropic-messages", "claude-opus-4-6"); +const vertexClaude = m("google-vertex", "anthropic-messages", "claude-opus-4-6"); +const gemini = m("google", "google-generative-ai", "gemini-3-flash"); +const vertexGemini = m("google-vertex", "google-vertex", "gemini-3-flash"); +const fireworks = m("fireworks", "openai-completions", "qwen3"); +const orOpenAI = m("openrouter", "openai-responses", "openai/gpt-5.5"); +const orGoogle = m("openrouter", "openai-completions", "google/gemini-3-flash"); +const orAnthropic = m("openrouter", "openai-completions", "anthropic/claude-opus-4-6"); + +describe("serviceTierFamily", () => { + it("classifies first-party providers by provider/api", () => { + expect(serviceTierFamily(openai)).toBe("openai"); + expect(serviceTierFamily(codex)).toBe("openai"); + expect(serviceTierFamily(anthropic)).toBe("anthropic"); + expect(serviceTierFamily(vertexClaude)).toBe("anthropic"); // Claude on Vertex is the anthropic family + expect(serviceTierFamily(gemini)).toBe("google"); + expect(serviceTierFamily(vertexGemini)).toBe("google"); + expect(serviceTierFamily(fireworks)).toBeUndefined(); }); - it("counts priority tier as one premium request on OpenAI Codex", () => { - expect(getPriorityPremiumRequests("priority", "openai-codex")).toBe(1); - }); - - it("ignores non-priority paid tiers", () => { - expect(getPriorityPremiumRequests("flex", "openai")).toBe(0); - expect(getPriorityPremiumRequests("scale", "openai")).toBe(0); - }); - - it("ignores default and auto tiers", () => { - expect(getPriorityPremiumRequests("default", "openai")).toBe(0); - expect(getPriorityPremiumRequests("auto", "openai")).toBe(0); - }); - - it("ignores priority tier on providers that drop service_tier", () => { - // `priority` is realized on `openai`, `openai-codex`, and direct `anthropic` - // (as fast mode). Everywhere else it's silently dropped, so it must not - // be billed as premium. - expect(getPriorityPremiumRequests("priority", "github-copilot")).toBe(0); - expect(getPriorityPremiumRequests("priority", "azure")).toBe(0); - expect(getPriorityPremiumRequests("priority", "bedrock")).toBe(0); - }); - - it("counts priority on direct Anthropic as one premium request (fast mode)", () => { - expect(getPriorityPremiumRequests("priority", "anthropic")).toBe(1); - }); - - it("returns zero when service tier is unset", () => { - expect(getPriorityPremiumRequests(undefined, "openai")).toBe(0); - expect(getPriorityPremiumRequests(null, "openai")).toBe(0); - }); - - describe("scoped tiers", () => { - it("treats `openai-only` as priority on OpenAI and OpenAI-Codex", () => { - expect(getPriorityPremiumRequests("openai-only", "openai")).toBe(1); - expect(getPriorityPremiumRequests("openai-only", "openai-codex")).toBe(1); - }); - - it("treats `openai-only` as inactive on Anthropic and everywhere else", () => { - expect(getPriorityPremiumRequests("openai-only", "anthropic")).toBe(0); - expect(getPriorityPremiumRequests("openai-only", "github-copilot")).toBe(0); - expect(getPriorityPremiumRequests("openai-only", "bedrock")).toBe(0); - }); - - it("treats `claude-only` as priority on direct Anthropic", () => { - expect(getPriorityPremiumRequests("claude-only", "anthropic")).toBe(1); - }); - - it("treats `claude-only` as inactive on OpenAI, Bedrock/Vertex, and elsewhere", () => { - expect(getPriorityPremiumRequests("claude-only", "openai")).toBe(0); - expect(getPriorityPremiumRequests("claude-only", "openai-codex")).toBe(0); - expect(getPriorityPremiumRequests("claude-only", "bedrock")).toBe(0); - expect(getPriorityPremiumRequests("claude-only", "vertex")).toBe(0); - }); + it("classifies OpenRouter models by id namespace", () => { + expect(serviceTierFamily(orOpenAI)).toBe("openai"); + expect(serviceTierFamily(orGoogle)).toBe("google"); + expect(serviceTierFamily(orAnthropic)).toBe("anthropic"); + expect(serviceTierFamily(m("openrouter", "openai-completions", "z-ai/glm-4.7"))).toBeUndefined(); }); }); -describe("resolveServiceTier", () => { - it("passes unscoped tiers through unchanged for any provider", () => { - expect(resolveServiceTier("flex", "openai")).toBe("flex"); - expect(resolveServiceTier("priority", "anthropic")).toBe("priority"); - expect(resolveServiceTier("auto", "openai-codex")).toBe("auto"); - expect(resolveServiceTier("default", "github-copilot")).toBe("default"); - }); - - it("scopes `openai-only` to OpenAI providers", () => { - expect(resolveServiceTier("openai-only", "openai")).toBe("priority"); - expect(resolveServiceTier("openai-only", "openai-codex")).toBe("priority"); - expect(resolveServiceTier("openai-only", "anthropic")).toBeUndefined(); - expect(resolveServiceTier("openai-only", "bedrock")).toBeUndefined(); - expect(resolveServiceTier("openai-only", undefined)).toBeUndefined(); - }); - - it("scopes `claude-only` to direct Anthropic", () => { - expect(resolveServiceTier("claude-only", "anthropic")).toBe("priority"); - expect(resolveServiceTier("claude-only", "openai")).toBeUndefined(); - expect(resolveServiceTier("claude-only", "bedrock")).toBeUndefined(); - expect(resolveServiceTier("claude-only", "vertex")).toBeUndefined(); - }); - - it("returns undefined for null/undefined input", () => { - expect(resolveServiceTier(undefined, "openai")).toBeUndefined(); - expect(resolveServiceTier(null, "openai")).toBeUndefined(); +describe("resolveModelServiceTier", () => { + it("reduces a per-family map to the model's family entry", () => { + const tiers = { openai: "priority", anthropic: "priority", google: "flex" } as const; + expect(resolveModelServiceTier(tiers, openai)).toBe("priority"); + expect(resolveModelServiceTier(tiers, gemini)).toBe("flex"); + expect(resolveModelServiceTier(tiers, orAnthropic)).toBe("priority"); + expect(resolveModelServiceTier(tiers, fireworks)).toBeUndefined(); // no family + expect(resolveModelServiceTier(undefined, openai)).toBeUndefined(); + expect(resolveModelServiceTier({ google: "priority" }, openai)).toBeUndefined(); }); }); describe("shouldSendServiceTier", () => { - it("returns false for non-OpenAI/non-Fireworks providers", () => { - expect(shouldSendServiceTier("flex", "azure-openai-responses")).toBe(false); - expect(shouldSendServiceTier("scale", "firepass")).toBe(false); + it("sends flex/scale/priority on the OpenAI family and OpenRouter", () => { + for (const p of ["openai", "openai-codex", "openrouter"]) { + expect(shouldSendServiceTier("flex", p)).toBe(true); + expect(shouldSendServiceTier("scale", p)).toBe(true); + expect(shouldSendServiceTier("priority", p)).toBe(true); + expect(shouldSendServiceTier("default", p)).toBe(false); + expect(shouldSendServiceTier("auto", p)).toBe(false); + } + }); + + it("sends flex/priority on direct Google, priority-only on Vertex (no scale)", () => { + expect(shouldSendServiceTier("flex", "google")).toBe(true); + expect(shouldSendServiceTier("priority", "google")).toBe(true); + expect(shouldSendServiceTier("scale", "google")).toBe(false); + expect(shouldSendServiceTier("priority", "google-vertex")).toBe(true); + expect(shouldSendServiceTier("flex", "google-vertex")).toBe(false); // Vertex flex has no wire control + }); + + it("sends only priority on Fireworks, nothing on Anthropic", () => { + expect(shouldSendServiceTier("priority", "fireworks")).toBe(true); + expect(shouldSendServiceTier("flex", "fireworks")).toBe(false); expect(shouldSendServiceTier("priority", "anthropic")).toBe(false); }); - it("returns true for fireworks only with the priority tier", () => { - expect(shouldSendServiceTier("priority", "fireworks")).toBe(true); - // Fireworks realizes only the Priority serving path — flex/scale are OpenAI-only. - expect(shouldSendServiceTier("flex", "fireworks")).toBe(false); - expect(shouldSendServiceTier("scale", "fireworks")).toBe(false); - expect(shouldSendServiceTier("auto", "fireworks")).toBe(false); - expect(shouldSendServiceTier("default", "fireworks")).toBe(false); - expect(shouldSendServiceTier(undefined, "fireworks")).toBe(false); - }); - - it("returns true for openai with priority/flex/scale tiers", () => { - expect(shouldSendServiceTier("priority", "openai")).toBe(true); - expect(shouldSendServiceTier("flex", "openai")).toBe(true); - expect(shouldSendServiceTier("scale", "openai")).toBe(true); - }); - - it("returns true for openai-codex with priority/flex/scale tiers", () => { - expect(shouldSendServiceTier("priority", "openai-codex")).toBe(true); - expect(shouldSendServiceTier("flex", "openai-codex")).toBe(true); - expect(shouldSendServiceTier("scale", "openai-codex")).toBe(true); - }); - - it("returns false for default tier on OpenAI providers", () => { - expect(shouldSendServiceTier("default", "openai")).toBe(false); - expect(shouldSendServiceTier("default", "openai-codex")).toBe(false); - }); - - it("returns false for auto tier on OpenAI providers", () => { - expect(shouldSendServiceTier("auto", "openai")).toBe(false); - expect(shouldSendServiceTier("auto", "openai-codex")).toBe(false); - }); - - it("returns false for undefined/null tier", () => { + it("returns false for unset tiers", () => { expect(shouldSendServiceTier(undefined, "openai")).toBe(false); expect(shouldSendServiceTier(null, "openai")).toBe(false); }); }); + +describe("realizesPriorityServiceTier", () => { + it("realizes priority where the wire actually applies it", () => { + expect(realizesPriorityServiceTier("priority", openai)).toBe(true); + expect(realizesPriorityServiceTier("priority", anthropic)).toBe(true); // direct fast mode + expect(realizesPriorityServiceTier("priority", gemini)).toBe(true); + expect(realizesPriorityServiceTier("priority", vertexGemini)).toBe(true); + expect(realizesPriorityServiceTier("priority", fireworks)).toBe(true); + expect(realizesPriorityServiceTier("priority", orOpenAI)).toBe(true); + expect(realizesPriorityServiceTier("priority", orGoogle)).toBe(true); + }); + + it("does not realize priority where the wire drops it", () => { + expect(realizesPriorityServiceTier("priority", vertexClaude)).toBe(false); // no fast mode on Vertex + expect(realizesPriorityServiceTier("priority", orAnthropic)).toBe(false); // OpenRouter Anthropic + expect(realizesPriorityServiceTier("flex", openai)).toBe(false); + expect(realizesPriorityServiceTier(undefined, openai)).toBe(false); + }); +}); + +describe("getPriorityPremiumRequests", () => { + it("counts one premium request per realized priority on billing providers", () => { + expect(getPriorityPremiumRequests("priority", openai)).toBe(1); + expect(getPriorityPremiumRequests("priority", codex)).toBe(1); + expect(getPriorityPremiumRequests("priority", anthropic)).toBe(1); + expect(getPriorityPremiumRequests("priority", gemini)).toBe(1); + expect(getPriorityPremiumRequests("priority", vertexGemini)).toBe(1); + }); + + it("does not bill OpenRouter, unrealized, or non-priority traffic", () => { + expect(getPriorityPremiumRequests("priority", orOpenAI)).toBe(0); // OpenRouter bills its own way + expect(getPriorityPremiumRequests("priority", vertexClaude)).toBe(0); // not realized + expect(getPriorityPremiumRequests("priority", fireworks)).toBe(0); // realized but not Copilot-premium + expect(getPriorityPremiumRequests("flex", openai)).toBe(0); + expect(getPriorityPremiumRequests(undefined, openai)).toBe(0); + }); +}); + +describe("coerceServiceTierByFamily", () => { + it("migrates legacy scalar values to a per-family map", () => { + expect(coerceServiceTierByFamily("priority")).toEqual({ + openai: "priority", + anthropic: "priority", + google: "priority", + }); + expect(coerceServiceTierByFamily("openai-only")).toEqual({ openai: "priority" }); + expect(coerceServiceTierByFamily("claude-only")).toEqual({ anthropic: "priority" }); + expect(coerceServiceTierByFamily("flex")).toEqual({ openai: "flex" }); + expect(coerceServiceTierByFamily("none")).toBeUndefined(); + expect(coerceServiceTierByFamily(null)).toBeUndefined(); + }); + + it("passes a per-family map through, dropping invalid entries", () => { + expect(coerceServiceTierByFamily({ openai: "priority", google: "flex" })).toEqual({ + openai: "priority", + google: "flex", + }); + expect(coerceServiceTierByFamily({ openai: "bogus" })).toBeUndefined(); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a8e76887c..7571e2a01 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,11 @@ ### Changed +- Replaced the global `serviceTier` setting with `tier.openai`, `tier.anthropic`, and `tier.google` for granular control +- Updated `/fast` to target the service-tier family of the currently selected model +- Updated subagent and advisor tier configuration to use the new per-family setting structure +- Removed `fastModeScope` setting, as per-family scoping is now natively supported via the `tier.*` settings + - Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files. - Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content. diff --git a/packages/coding-agent/src/cli/bench-cli.ts b/packages/coding-agent/src/cli/bench-cli.ts index 52ed29b28..9e8a0a537 100644 --- a/packages/coding-agent/src/cli/bench-cli.ts +++ b/packages/coding-agent/src/cli/bench-cli.ts @@ -10,9 +10,10 @@ import type { Model, ProviderSessionState, ServiceTier, + ServiceTierByFamily, SimpleStreamOptions, } from "@oh-my-pi/pi-ai"; -import { streamSimple } from "@oh-my-pi/pi-ai"; +import { resolveModelServiceTier, streamSimple } from "@oh-my-pi/pi-ai"; import { buildModelProviderPriorityRank, type CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity"; import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui"; import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils"; @@ -25,7 +26,7 @@ import { getModelMatchPreferences, resolveCliModel, } from "../config/model-resolver"; -import { resolveServiceTierSetting } from "../config/service-tier"; +import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; import { Settings } from "../config/settings"; import benchPrompt from "../prompts/bench.md" with { type: "text" }; import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk"; @@ -106,8 +107,8 @@ export interface BenchSummary { maxTokens: number; models: BenchModelReport[]; failures: number; - /** Requested service tier passed to every request; absent when none was requested. Scoped tiers (`openai-only`/`claude-only`) may be dropped per-provider downstream. */ - serviceTier?: ServiceTier; + /** Requested per-family service tiers, resolved per model before reaching the wire. */ + serviceTierByFamily?: ServiceTierByFamily; } type BenchStreamSimple = ( @@ -518,12 +519,18 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe const runtime = await (deps.createRuntime ?? createDefaultRuntime)(); try { const targets = resolveBenchModels(command.models, runtime.modelRegistry, runtime.settings, writeStderr); - // Explicit `--service-tier` wins; otherwise fall back to the configured - // `serviceTier` setting (`none`/unset omits the wire field). Scope-aware - // gating to the model's provider happens downstream in the provider layer. - const serviceTierValue = command.flags.serviceTier ?? runtime.settings?.get("serviceTier"); - const serviceTier = serviceTierValue ? resolveServiceTierSetting(serviceTierValue, undefined) : undefined; - if (!json && serviceTier) writeStdout(`${chalk.dim(`service tier: ${serviceTier}`)}\n`); + // Explicit `--service-tier` (a single value broadcast across families) wins; + // otherwise fall back to the configured per-family `tier.*` settings. Each + // model resolves its own family's tier below before reaching the wire. + const flagTier = command.flags.serviceTier ? serviceTierSettingToTier(command.flags.serviceTier) : undefined; + const serviceTierByFamily = command.flags.serviceTier + ? serviceTierForAllFamilies(flagTier) + : buildServiceTierByFamily( + runtime.settings?.get("tier.openai") ?? "none", + runtime.settings?.get("tier.anthropic") ?? "none", + runtime.settings?.get("tier.google") ?? "none", + ); + if (!json && flagTier) writeStdout(`${chalk.dim(`service tier: ${flagTier}`)}\n`); const reports: BenchModelReport[] = []; for (const { selector, model, thinking } of targets) { if (!json) { @@ -564,7 +571,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe maxTokens, reasoning: toReasoningEffort(thinking), disableReasoning: shouldDisableReasoning(thinking) ? true : undefined, - serviceTier, + serviceTier: resolveModelServiceTier(serviceTierByFamily, model), }, streamFn, now, @@ -606,7 +613,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe reports.push(buildModelReport(selector, model, thinking, results)); } const failures = reports.reduce((sum, report) => sum + report.results.filter(result => !result.ok).length, 0); - const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTier }; + const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTierByFamily }; if (json) { writeStdout(`${JSON.stringify(summary, null, 2)}\n`); } else if (reports.length > 1 || runs > 1) { diff --git a/packages/coding-agent/src/commands/bench.ts b/packages/coding-agent/src/commands/bench.ts index 20d0a3a04..606530377 100644 --- a/packages/coding-agent/src/commands/bench.ts +++ b/packages/coding-agent/src/commands/bench.ts @@ -1,6 +1,6 @@ import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; import { runBenchCommand } from "../cli/bench-cli"; -import { SERVICE_TIER_SETTING_VALUES } from "../config/service-tier"; +import { SERVICE_TIER_OPENAI_VALUES } from "../config/service-tier"; export default class Bench extends Command { static description = @@ -19,8 +19,8 @@ export default class Bench extends Command { "max-tokens": Flags.integer({ description: "Max output tokens per request", default: 512 }), prompt: Flags.string({ description: "Custom prompt text (default: bundled bench prompt)" }), "service-tier": Flags.string({ - description: "Service tier hint (default: configured `serviceTier` setting; `none` omits it)", - options: SERVICE_TIER_SETTING_VALUES, + description: "Service tier applied per model family (default: configured `tier.*` settings; `none` omits it)", + options: SERVICE_TIER_OPENAI_VALUES, }), json: Flags.boolean({ description: "Output JSON" }), par: Flags.integer({ description: "Execute runs with N parallel queries/requests", default: 4 }), diff --git a/packages/coding-agent/src/config/service-tier.ts b/packages/coding-agent/src/config/service-tier.ts index 807343c19..74ce6d1f7 100644 --- a/packages/coding-agent/src/config/service-tier.ts +++ b/packages/coding-agent/src/config/service-tier.ts @@ -1,87 +1,116 @@ -import type { ServiceTier } from "@oh-my-pi/pi-ai"; +import type { ServiceTier, ServiceTierByFamily } from "@oh-my-pi/pi-ai"; import type { SubmenuOption } from "./settings-schema"; /** - * Service-tier setting values shared by every "Service Tier" setting. `"none"` - * is the omit-the-parameter sentinel; the remaining values mirror - * {@link ServiceTier}. + * Per-family service-tier setting values. `"none"` is the omit-the-parameter + * sentinel; the rest mirror the wire {@link ServiceTier} values each provider + * family actually realizes. OpenAI accepts the full set; Anthropic realizes + * only `priority` (fast mode); Google (Gemini API + Vertex) realizes + * `flex`/`priority`. */ -export const SERVICE_TIER_SETTING_VALUES = [ +export const SERVICE_TIER_OPENAI_VALUES = ["none", "auto", "default", "flex", "scale", "priority"] as const; +export const SERVICE_TIER_ANTHROPIC_VALUES = ["none", "priority"] as const; +export const SERVICE_TIER_GOOGLE_VALUES = ["none", "flex", "priority"] as const; + +export type ServiceTierOpenAISettingValue = (typeof SERVICE_TIER_OPENAI_VALUES)[number]; +export type ServiceTierAnthropicSettingValue = (typeof SERVICE_TIER_ANTHROPIC_VALUES)[number]; +export type ServiceTierGoogleSettingValue = (typeof SERVICE_TIER_GOOGLE_VALUES)[number]; + +/** + * Inherit-capable single value for the subagent/advisor tiers. The chosen tier + * is broadcast across families and applied to whichever family the spawned + * model belongs to (clamped to what that family realizes); `"inherit"` defers + * to the main agent's live per-family selection. + */ +export const SERVICE_TIER_INHERIT_SETTING_VALUES = [ + "inherit", "none", "auto", "default", "flex", "scale", "priority", - "openai-only", - "claude-only", ] as const; -export type ServiceTierSettingValue = (typeof SERVICE_TIER_SETTING_VALUES)[number]; - -/** Variant value set for scoped service-tier settings (subagent/advisor) that can defer to the main agent. */ -export const SERVICE_TIER_INHERIT_SETTING_VALUES = ["inherit", ...SERVICE_TIER_SETTING_VALUES] as const; - export type ServiceTierInheritSettingValue = (typeof SERVICE_TIER_INHERIT_SETTING_VALUES)[number]; -/** Submenu descriptions shared by the base `serviceTier` setting. */ -export const SERVICE_TIER_OPTIONS: ReadonlyArray> = [ - { value: "none", label: "None", description: "Omit service_tier parameter" }, - { value: "auto", label: "Auto", description: "Use provider default tier selection (OpenAI)" }, - { value: "default", label: "Default", description: "Standard priority processing (OpenAI)" }, - { value: "flex", label: "Flex", description: "Flexible capacity tier when available (OpenAI)" }, - { value: "scale", label: "Scale", description: "Scale Tier credits when available (OpenAI)" }, +export const SERVICE_TIER_OPENAI_OPTIONS: ReadonlyArray> = [ + { value: "none", label: "None", description: "Omit service_tier (standard processing)" }, + { value: "auto", label: "Auto", description: "Provider default tier selection" }, + { value: "default", label: "Default", description: "Standard priority processing" }, + { value: "flex", label: "Flex", description: "Lower cost, higher latency when available" }, + { value: "scale", label: "Scale", description: "Scale Tier credits when available" }, + { value: "priority", label: "Priority", description: "Faster, higher cost (premium request)" }, +]; + +export const SERVICE_TIER_ANTHROPIC_OPTIONS: ReadonlyArray> = [ + { value: "none", label: "None", description: "Standard processing" }, { value: "priority", label: "Priority", - description: "Priority on every supported provider (OpenAI `service_tier`, Anthropic fast mode)", - }, - { - value: "openai-only", - label: "Priority (OpenAI only)", - description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere", - }, - { - value: "claude-only", - label: "Priority (Claude only)", - description: "Anthropic fast mode on direct Claude requests; ignored elsewhere (incl. Bedrock/Vertex)", + description: 'Fast mode (`speed: "fast"`) on supported direct Claude models; ignored on Bedrock/Vertex', }, ]; -/** Submenu descriptions for inherit-capable service-tier settings. */ +export const SERVICE_TIER_GOOGLE_OPTIONS: ReadonlyArray> = [ + { value: "none", label: "None", description: "Standard processing" }, + { value: "flex", label: "Flex", description: "Lower cost, higher latency (Gemini API + Vertex)" }, + { value: "priority", label: "Priority", description: "Faster, higher reliability (Gemini API + Vertex)" }, +]; + export const SERVICE_TIER_INHERIT_OPTIONS: ReadonlyArray> = [ - { value: "inherit", label: "Inherit", description: "Use the main agent's Service Tier" }, - ...SERVICE_TIER_OPTIONS, + { value: "inherit", label: "Inherit", description: "Match the main agent's live per-family tiers" }, + { value: "none", label: "None", description: "Standard processing" }, + { value: "auto", label: "Auto", description: "Provider default tier selection (OpenAI family)" }, + { value: "default", label: "Default", description: "Standard priority processing (OpenAI family)" }, + { value: "flex", label: "Flex", description: "Flexible capacity tier (OpenAI/Google families)" }, + { value: "scale", label: "Scale", description: "Scale Tier credits (OpenAI family)" }, + { value: "priority", label: "Priority", description: "Priority on every supported family of the spawned model" }, ]; -/** - * Resolve a service-tier setting value to the wire {@link ServiceTier} (or - * `undefined` to omit). `"inherit"` defers to `inherited`; `"none"` omits. - */ -export function resolveServiceTierSetting(value: string, inherited: ServiceTier | undefined): ServiceTier | undefined { - if (value === "inherit") return inherited; - if (value === "none" || value === "") return undefined; +/** Map a per-family setting value to a wire {@link ServiceTier}, or `undefined` to omit. */ +export function serviceTierSettingToTier(value: string): ServiceTier | undefined { + if (value === "none" || value === "" || value === "inherit") return undefined; return value as ServiceTier; } -/** - * Resolve the `serviceTier` *setting value* to stamp onto a subagent's settings - * snapshot. - * - * - A concrete `subagentSetting` (`"none"` or a tier) wins outright. - * - `"inherit"` defers to the parent's live effective tier when the caller has a - * live session (`inherited` passed as `ServiceTier | null`, where `null` means - * the parent explicitly has no tier — e.g. `/fast off`). When no live session - * is available (`inherited === undefined`, e.g. cold subagent revive) it falls - * back to the parent's configured `serviceTier` setting so behavior matches a - * plain settings snapshot. - */ -export function resolveSubagentServiceTier( - subagentSetting: string, - configuredTier: ServiceTierSettingValue, - inherited: ServiceTier | null | undefined, -): ServiceTierSettingValue { - if (subagentSetting !== "inherit") return subagentSetting as ServiceTierSettingValue; - if (inherited === undefined) return configuredTier; - return inherited ?? "none"; +/** Assemble the live per-family tier map from the three `tier.*` setting values. */ +export function buildServiceTierByFamily(openai: string, anthropic: string, google: string): ServiceTierByFamily { + const out: ServiceTierByFamily = {}; + const o = serviceTierSettingToTier(openai); + if (o) out.openai = o; + const a = serviceTierSettingToTier(anthropic); + if (a) out.anthropic = a; + const g = serviceTierSettingToTier(google); + if (g) out.google = g; + return out; +} + +/** + * Broadcast a single chosen tier across families, clamped to what each family + * realizes: OpenAI takes any tier, Anthropic only `priority`, Google only + * `flex`/`priority`. Used by the subagent/advisor single-value settings and the + * `omp bench --service-tier` flag, which apply one tier to whatever family the + * target model belongs to. + */ +export function serviceTierForAllFamilies(tier: ServiceTier | undefined): ServiceTierByFamily { + if (!tier) return {}; + const out: ServiceTierByFamily = { openai: tier }; + if (tier === "priority") out.anthropic = "priority"; + if (tier === "flex" || tier === "priority") out.google = tier; + return out; +} + +/** + * Resolve a subagent/advisor service-tier setting to a per-family map. + * + * - A concrete tier is broadcast across families (see + * {@link serviceTierForAllFamilies}). + * - `"none"` yields an empty map. + * - `"inherit"` defers to `inherited` — the parent's live per-family tiers when + * a live session supplied them, else the empty map. + */ +export function resolveSubagentServiceTier(setting: string, inherited: ServiceTierByFamily): ServiceTierByFamily { + if (setting === "inherit") return inherited; + return serviceTierForAllFamilies(serviceTierSettingToTier(setting)); } diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 14888734e..da1514818 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -36,10 +36,14 @@ import { import { EDIT_MODES } from "../utils/edit-mode"; import { SEARCH_PROVIDER_OPTIONS, SEARCH_PROVIDER_PREFERENCES, type SearchProviderId } from "../web/search/types"; import { + SERVICE_TIER_ANTHROPIC_OPTIONS, + SERVICE_TIER_ANTHROPIC_VALUES, + SERVICE_TIER_GOOGLE_OPTIONS, + SERVICE_TIER_GOOGLE_VALUES, SERVICE_TIER_INHERIT_OPTIONS, SERVICE_TIER_INHERIT_SETTING_VALUES, - SERVICE_TIER_OPTIONS, - SERVICE_TIER_SETTING_VALUES, + SERVICE_TIER_OPENAI_OPTIONS, + SERVICE_TIER_OPENAI_VALUES, } from "./service-tier"; /** Unified settings schema - single source of truth for all settings. @@ -1214,75 +1218,77 @@ export const SETTINGS_SCHEMA = { }, }, - serviceTier: { + "tier.openai": { type: "enum", - values: SERVICE_TIER_SETTING_VALUES, + values: SERVICE_TIER_OPENAI_VALUES, default: "none", ui: { tab: "model", group: "Sampling", - label: "Service Tier", + label: "Service Tier — OpenAI", description: - 'Processing priority hint (none = omit). OpenAI accepts the tier values directly; Anthropic realizes `priority` as `speed: "fast"` on supported Opus models. Scoped values target one family.', - options: SERVICE_TIER_OPTIONS, + "Processing tier for OpenAI / OpenAI-Codex requests, and OpenAI-family models routed via OpenRouter (none = omit). Sent as `service_tier`.", + options: SERVICE_TIER_OPENAI_OPTIONS, }, }, - serviceTierSubagent: { + "tier.anthropic": { + type: "enum", + values: SERVICE_TIER_ANTHROPIC_VALUES, + default: "none", + ui: { + tab: "model", + group: "Sampling", + label: "Service Tier — Anthropic", + description: + 'Processing tier for Claude requests. `priority` realizes fast mode (`speed: "fast"`) on supported direct Anthropic models; ignored on Bedrock/Vertex Claude and via OpenRouter.', + options: SERVICE_TIER_ANTHROPIC_OPTIONS, + }, + }, + + "tier.google": { + type: "enum", + values: SERVICE_TIER_GOOGLE_VALUES, + default: "none", + ui: { + tab: "model", + group: "Sampling", + label: "Service Tier — Google", + description: + "Processing tier for Gemini (Google AI Studio + Vertex) requests, and Google-family models routed via OpenRouter (none = omit). Sent as the top-level `serviceTier` field.", + options: SERVICE_TIER_GOOGLE_OPTIONS, + }, + }, + + "tier.subagent": { type: "enum", values: SERVICE_TIER_INHERIT_SETTING_VALUES, default: "inherit", ui: { tab: "model", group: "Sampling", - label: "Service Tier - Subagent", + label: "Service Tier — Subagent", description: - "Service Tier for spawned task/eval subagents. Inherit = match the main agent's live tier (tracks /fast); pick a value to scope subagents independently.", + "Service Tier for spawned task/eval subagents. Inherit = match the main agent's live per-family tiers (tracks /fast); pick a value to apply it to whichever family the subagent's model belongs to.", options: SERVICE_TIER_INHERIT_OPTIONS, }, }, - serviceTierAdvisor: { + "tier.advisor": { type: "enum", values: SERVICE_TIER_INHERIT_SETTING_VALUES, default: "none", ui: { tab: "model", group: "Sampling", - label: "Service Tier - Advisor", + label: "Service Tier — Advisor", description: - "Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live tier; pick a value (e.g. Priority) to run the advisor on a faster serving path.", + "Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live per-family tiers; pick a value to apply it to the advisor model's family.", options: SERVICE_TIER_INHERIT_OPTIONS, condition: "advisorEnabled", }, }, - fastModeScope: { - type: "enum", - values: ["both", "openai", "claude"] as const, - default: "both", - ui: { - tab: "model", - group: "Sampling", - label: "Fast Mode Scope", - description: - 'Which providers `/fast on` (and the fast-mode toggle) target. "both" = priority on every supported provider; "openai"/"claude" scope it to one family (mirrors serviceTier openai-only/claude-only).', - options: [ - { value: "both", label: "Both", description: "Priority on every supported provider" }, - { - value: "openai", - label: "OpenAI only", - description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere", - }, - { - value: "claude", - label: "Claude only", - description: "Anthropic fast mode on direct Claude requests; ignored elsewhere", - }, - ], - }, - }, - // Retries "retry.enabled": { type: "boolean", default: true }, diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index 0b2155aed..b517105e3 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -1148,6 +1148,53 @@ export class Settings { // the incoherent "hashline edits without addressable anchors" state. delete raw.readHashLines; + // serviceTier (single enum with scoped openai-only/claude-only sentinels) + // → per-family tier.openai/tier.anthropic/tier.google; serviceTierSubagent + // → tier.subagent; serviceTierAdvisor → tier.advisor. `fastModeScope` is + // dropped — per-family scoping is now expressed by the three tier settings. + const tierObj = isRecord(raw.tier) ? raw.tier : {}; + let tierTouched = false; + const setTier = (family: string, value: unknown): void => { + if (value !== undefined && !(family in tierObj)) { + tierObj[family] = value; + tierTouched = true; + } + }; + if (typeof raw.serviceTier === "string") { + switch (raw.serviceTier) { + case "priority": + setTier("openai", "priority"); + setTier("anthropic", "priority"); + setTier("google", "priority"); + break; + case "openai-only": + setTier("openai", "priority"); + break; + case "claude-only": + setTier("anthropic", "priority"); + break; + case "auto": + case "default": + case "flex": + case "scale": + setTier("openai", raw.serviceTier); + break; + } + delete raw.serviceTier; + } + const mapInheritTier = (value: unknown): unknown => + value === "openai-only" || value === "claude-only" ? "priority" : value; + if ("serviceTierSubagent" in raw) { + setTier("subagent", mapInheritTier(raw.serviceTierSubagent)); + delete raw.serviceTierSubagent; + } + if ("serviceTierAdvisor" in raw) { + setTier("advisor", mapInheritTier(raw.serviceTierAdvisor)); + delete raw.serviceTierAdvisor; + } + if (tierTouched) raw.tier = tierObj; + delete raw.fastModeScope; + return raw; } diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index 0a2232421..301709281 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -405,8 +405,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption parentMnemopiSessionState: options.session.getMnemopiSessionState?.(), parentTelemetry: options.session.getTelemetry?.(), parentAgentId: options.session.getAgentId?.() ?? MAIN_AGENT_ID, - // Live source of truth for `serviceTierSubagent: inherit` (null = explicit none). - parentServiceTier: options.session.getServiceTier ? (options.session.getServiceTier() ?? null) : undefined, + // Live source of truth for `tier.subagent: inherit` (null = explicit none). + parentServiceTier: options.session.getServiceTierByFamily + ? (options.session.getServiceTierByFamily() ?? null) + : undefined, // Deliberately omit parentEvalSessionId: the parent's Python kernel is // blocked on this bridge call, so sharing the eval session would deadlock // (subagent queues behind the parent's in-flight execution, parent waits diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index a39cbf972..d75545663 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -140,7 +140,7 @@ const HOST_DEFAULTED_SETTING_PATHS: SettingPath[] = [ "advisor.subagents", "advisor.syncBacklog", "advisor.immuneTurns", - "serviceTierAdvisor", + "tier.advisor", ]; const RPC_BACKGROUND_DEFAULTED_SETTING_PATHS: SettingPath[] = [ diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index f43483950..7ac180663 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -42,6 +42,7 @@ import { resolveModelRoleValue, } from "./config/model-resolver"; import { loadPromptTemplates as loadPromptTemplatesInternal, type PromptTemplate } from "./config/prompt-templates"; +import { buildServiceTierByFamily } from "./config/service-tier"; import { Settings, type SkillsSettings } from "./config/settings"; import { CursorExecHandlers } from "./cursor"; import "./discovery"; @@ -1537,7 +1538,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} getModelString: () => (hasExplicitModel && model ? formatModelString(model) : undefined), getActiveModelString, getActiveModel: () => agent?.state.model ?? model, - getServiceTier: () => session?.serviceTier, + getServiceTierByFamily: () => session?.serviceTierByFamily, getImageAttachments: () => session?.getImageAttachments() ?? [], getPlanModeState: () => session?.getPlanModeState(), getPlanReferencePath: () => session?.getPlanReferencePath() ?? "local://PLAN.md", @@ -2525,13 +2526,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const openaiWebsocketSetting = settings.get("providers.openaiWebsockets") ?? "off"; const preferOpenAICodexWebsockets = openaiWebsocketSetting === "on" ? true : openaiWebsocketSetting === "off" ? false : undefined; - const serviceTierSetting = settings.get("serviceTier"); - - const initialServiceTier = hasServiceTierEntry - ? existingSession.serviceTier - : serviceTierSetting === "none" - ? undefined - : serviceTierSetting; + const initialServiceTierByFamily = hasServiceTierEntry + ? (existingSession.serviceTier ?? {}) + : buildServiceTierByFamily( + settings.get("tier.openai"), + settings.get("tier.anthropic"), + settings.get("tier.google"), + ); // One-shot launch-latency marker: fired the first time the loop dispatches // a chat request to the provider transport. See onFirstChatDispatch. @@ -2579,7 +2580,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} minP: settings.get("minP") >= 0 ? settings.get("minP") : undefined, presencePenalty: settings.get("presencePenalty") >= 0 ? settings.get("presencePenalty") : undefined, repetitionPenalty: settings.get("repetitionPenalty") >= 0 ? settings.get("repetitionPenalty") : undefined, - serviceTier: initialServiceTier, hideThinkingSummary: settings.get("omitThinking"), kimiApiFormat: settings.get("providers.kimiApiFormat") ?? "anthropic", preferWebsockets: preferOpenAICodexWebsockets, @@ -2639,8 +2639,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // classification persists its concrete effort once a real user turn runs. sessionManager.appendThinkingLevelChange(effectiveThinkingLevel); } - if (initialServiceTier) { - sessionManager.appendServiceTierChange(initialServiceTier); + if (Object.keys(initialServiceTierByFamily).length > 0) { + sessionManager.appendServiceTierChange(initialServiceTierByFamily); } } @@ -2691,6 +2691,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} agent, pruneToolDescriptions: inlineToolDescriptors, thinkingLevel: autoThinking ? AUTO_THINKING : effectiveThinkingLevel, + serviceTierByFamily: initialServiceTierByFamily, sessionManager, settings, autoApprove: options.autoApprove, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index bd16310c8..3e6a2bc6b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -92,6 +92,8 @@ import type { ResetCreditRedeemOutcome, ResetCreditTarget, ServiceTier, + ServiceTierByFamily, + ServiceTierFamily, SimpleStreamOptions, TextContent, ToolCall, @@ -105,7 +107,9 @@ import { deriveClaudeDeviceId, Effort, parseRateLimitReason, - resolveServiceTier, + realizesPriorityServiceTier, + resolveModelServiceTier, + serviceTierFamily, streamSimple, } from "@oh-my-pi/pi-ai"; import * as AIError from "@oh-my-pi/pi-ai/error"; @@ -168,7 +172,7 @@ import { } from "../config/model-resolver"; import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles"; import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates"; -import { resolveServiceTierSetting } from "../config/service-tier"; +import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; import type { Settings, SkillsSettings } from "../config/settings"; import { getDefault, onAppendOnlyModeChanged, validateProviderMaxInFlightRequests } from "../config/settings"; import { RawSseDebugBuffer } from "../debug/raw-sse-buffer"; @@ -506,6 +510,8 @@ export interface AgentSessionConfig { scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; /** Initial session thinking selector. */ thinkingLevel?: ConfiguredThinkingLevel; + /** Initial per-family service tiers (OpenAI / Anthropic / Google) for the live session. */ + serviceTierByFamily?: ServiceTierByFamily; /** Prompt templates for expansion */ promptTemplates?: PromptTemplate[]; /** File-based slash commands for expansion */ @@ -1787,6 +1793,7 @@ export class AgentSession { // toggle scopes priority to Fireworks alone, without mutating the shared // session `serviceTier` that drives `/fast` and OpenAI/Anthropic priority. this.agent.serviceTierResolver = model => this.#effectiveServiceTier(model); + this.#serviceTierByFamily = config.serviceTierByFamily ?? {}; this.#advisorTools = config.advisorTools; this.#advisorWatchdogPrompt = config.advisorWatchdogPrompt; this.#advisorSharedInstructions = config.advisorSharedInstructions; @@ -2045,15 +2052,20 @@ export class AgentSession { const legacy = !this.#advisorConfigs?.length; const roster: AdvisorConfig[] = legacy ? [{ name: "default" }] : this.#advisorConfigs!; - // Advisor service tier (`serviceTierAdvisor`): "none" (default) runs the - // advisor on standard processing; "inherit" tracks the session's live tier - // per request (like the main agent, including /fast toggles) via a resolver; - // a concrete value pins the advisor to that tier. One value for all advisors. - const advisorTierSetting = this.settings.get("serviceTierAdvisor"); - const advisorServiceTier = - advisorTierSetting === "inherit" ? undefined : resolveServiceTierSetting(advisorTierSetting, undefined); - const advisorServiceTierResolver = - advisorTierSetting === "inherit" ? (model: Model) => this.#effectiveServiceTier(model) : undefined; + // Advisor service tier (`tier.advisor`): "none" (default) runs the advisor + // on standard processing; "inherit" tracks the session's live per-family + // tiers per request (like the main agent, including /fast toggles); a + // concrete value is broadcast across families and applied to the advisor + // model's family. One value for all advisors. + const advisorTierSetting = this.settings.get("tier.advisor"); + const advisorTierMap = + advisorTierSetting === "inherit" + ? undefined + : serviceTierForAllFamilies(serviceTierSettingToTier(advisorTierSetting)); + const advisorServiceTierResolver = (model: Model): ServiceTier | undefined => + advisorTierSetting === "inherit" + ? this.#effectiveServiceTier(model) + : resolveModelServiceTier(advisorTierMap, model); const usedSlugs = new Set(); for (const config of roster) { @@ -2152,7 +2164,7 @@ export class AgentSession { transformProviderContext: this.#transformProviderContext, intentTracing: false, telemetry: advisorTelemetry, - serviceTier: advisorServiceTier, + serviceTier: undefined, serviceTierResolver: advisorServiceTierResolver, }); advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel)); @@ -3215,10 +3227,11 @@ export class AgentSession { if (event.message.role === "assistant") { this.#lastAssistantMessage = event.message; const assistantMsg = event.message as AssistantMessage; - const currentGrantsAnthropicPriority = - this.serviceTier === "priority" || this.serviceTier === "claude-only"; - if (assistantMsg.disabledFeatures?.includes("priority") && currentGrantsAnthropicPriority) { - this.setServiceTier(undefined); + if ( + assistantMsg.disabledFeatures?.includes("priority") && + this.#serviceTierByFamily.anthropic === "priority" + ) { + this.setServiceTierFamily("anthropic", undefined); this.emitNotice( "warning", "Priority/fast mode rejected for this model; retried without it. Fast mode is now off.", @@ -5203,8 +5216,11 @@ export class AgentSession { return this.#autoResolvedLevel; } - get serviceTier(): ServiceTier | undefined { - return this.agent.serviceTier; + #serviceTierByFamily: ServiceTierByFamily = {}; + + /** Live per-family service tiers (OpenAI / Anthropic / Google). */ + get serviceTierByFamily(): ServiceTierByFamily { + return this.#serviceTierByFamily; } /** Whether agent is currently streaming a response */ @@ -7892,7 +7908,7 @@ export class AgentSession { this.#scheduledHiddenNextTurnGeneration = undefined; this.sessionManager.appendThinkingLevelChange(this.thinkingLevel, this.configuredThinkingLevel()); - this.sessionManager.appendServiceTierChange(this.serviceTier ?? null); + this.sessionManager.appendServiceTierChange(this.#serviceTierEntry()); if (nextDiscoverySessionToolNames) { await this.#applyActiveToolsByName(nextDiscoverySessionToolNames, { persistMCPSelection: false }); if (this.getSelectedMCPToolNames().length > 0) { @@ -8434,38 +8450,36 @@ export class AgentSession { } /** - * True when *any* fast-mode-granting service tier is configured, regardless - * of whether the active model's provider actually realizes it. Used by the - * toggle (`/fast on|off`) so re-toggling a scoped tier (`openai-only`, - * `claude-only`) doesn't silently broaden it to unscoped `priority`. + * True when the currently selected model's family is set to `priority` — the + * `/fast` on/off state for the active model. Returns false when no model is + * selected or the model exposes no service-tier family (e.g. Fireworks, which + * has its own Providers › Fireworks Tier toggle). * - * For "is fast mode actually applied to the next request?" use - * {@link isFastModeActive} instead — that one respects the model's provider. + * For "is priority actually applied to the next request?" use + * {@link isFastModeActive} instead. */ isFastModeEnabled(): boolean { - return ( - this.serviceTier === "priority" || this.serviceTier === "claude-only" || this.serviceTier === "openai-only" - ); + const family = this.model ? serviceTierFamily(this.model) : undefined; + return family ? this.#serviceTierByFamily[family] === "priority" : false; } /** - * True when the configured `serviceTier` resolves to `"priority"` for the - * *currently selected model's provider*. Returns false for scoped tiers - * that don't match (e.g. `"openai-only"` on an anthropic model) and when - * no model is selected. + * True when `priority` is actually realized on the wire for the currently + * selected model (OpenAI/Google `service_tier`, direct Anthropic fast mode, + * or Fireworks priority). Returns false for tiers the active model can't + * realize and when no model is selected. */ isFastModeActive(): boolean { - return resolveServiceTier(this.#effectiveServiceTier(), this.model?.provider) === "priority"; + const model = this.model; + return !!model && realizesPriorityServiceTier(this.#effectiveServiceTier(model), model); } /** - * Effective wire service-tier for a request to `model`. Fireworks models - * take the Priority serving path only when the Providers › Fireworks Tier - * setting is `"priority"` — that toggle is the sole opt-in, so a global - * `serviceTier: "priority"` (for OpenAI/Anthropic) never silently incurs - * Fireworks priority costs — and never for `-fast` variants, whose Fast - * serving path is mutually exclusive with Priority. Every other provider - * uses the session `serviceTier` unchanged. + * Effective wire service-tier for a request to `model`. Fireworks models take + * the Priority serving path only when the Providers › Fireworks Tier setting + * is `"priority"` (and never for `-fast` variants, whose Fast serving path is + * mutually exclusive with Priority). Every other model resolves the live + * per-family tier map down to the entry for its family. */ #effectiveServiceTier(model: Model | undefined = this.model): ServiceTier | undefined { if (model?.provider === "fireworks") { @@ -8473,40 +8487,56 @@ export class AgentSession { ? "priority" : undefined; } - return this.serviceTier; + if (!model) return undefined; + return resolveModelServiceTier(this.#serviceTierByFamily, model); } - setServiceTier(serviceTier: ServiceTier | undefined): void { - if (this.serviceTier === serviceTier) return; - // Re-arming priority on Anthropic? Clear the per-session auto-fallback - // sticky disable so the next request actually carries `speed: "fast"` - // again. Without this, `/fast on` (or user switching to a tier that - // grants anthropic priority) after an auto-disable is a silent no-op - // and the warning notice fires every turn. - if (serviceTier === "priority" || serviceTier === "claude-only") { + /** The live per-family tier map, or `null` when empty (for session persistence). */ + #serviceTierEntry(): ServiceTierByFamily | null { + return Object.keys(this.#serviceTierByFamily).length > 0 ? this.#serviceTierByFamily : null; + } + + /** Set one family's tier (or clear it with `undefined`); persists the change. */ + setServiceTierFamily(family: ServiceTierFamily, tier: ServiceTier | undefined): void { + if (this.#serviceTierByFamily[family] === tier) return; + const next: ServiceTierByFamily = { ...this.#serviceTierByFamily }; + if (tier) next[family] = tier; + else delete next[family]; + this.#applyServiceTierByFamily(next); + } + + /** Replace the whole per-family tier map; persists + re-arms Anthropic fast mode. */ + #applyServiceTierByFamily(next: ServiceTierByFamily): void { + // Re-arming Anthropic priority clears the per-session fast-mode auto-disable + // so the next request actually carries `speed: "fast"` again. + if (next.anthropic === "priority" && this.#serviceTierByFamily.anthropic !== "priority") { clearAnthropicFastModeFallback(this.#providerSessionState); } - this.agent.serviceTier = serviceTier; - this.sessionManager.appendServiceTierChange(serviceTier ?? null); + this.#serviceTierByFamily = next; + this.sessionManager.appendServiceTierChange(this.#serviceTierEntry()); } + /** + * `/fast on|off` targets the family of the currently selected model: it sets + * (or clears) that family's `priority` tier. Models without a service-tier + * family (Fireworks, or providers with no tier knob) have nothing to toggle. + */ setFastMode(enabled: boolean): void { - if (enabled && this.isFastModeEnabled()) { - // Already on under any scope — keep the user's scoped value. + const family = this.model ? serviceTierFamily(this.model) : undefined; + if (!family) { + this.emitNotice("info", "The current model has no service-tier control for /fast to toggle.", "priority"); return; } if (!enabled) { - this.setServiceTier(undefined); + if (this.#serviceTierByFamily[family] === "priority") this.setServiceTierFamily(family, undefined); return; } - const scope = this.settings.get("fastModeScope"); - this.setServiceTier(scope === "openai" ? "openai-only" : scope === "claude" ? "claude-only" : "priority"); + this.setServiceTierFamily(family, "priority"); } toggleFastMode(): boolean { - const enabled = !this.isFastModeEnabled(); - this.setFastMode(enabled); - return enabled; + this.setFastMode(!this.isFastModeEnabled()); + return this.isFastModeEnabled(); } /** @@ -13236,7 +13266,7 @@ export class AgentSession { const previousThinkingLevel = this.#thinkingLevel; const previousAutoThinking = this.#autoThinking; const previousAutoResolvedLevel = this.#autoResolvedLevel; - const previousServiceTier = this.agent.serviceTier; + const previousServiceTierByFamily = this.#serviceTierByFamily; const previousSelectedMCPToolNames = new Set(this.#selectedMCPToolNames); const previousTools = [...this.agent.state.tools]; const previousBaseSystemPrompt = this.#baseSystemPrompt; @@ -13322,7 +13352,11 @@ export class AgentSession { .getBranch() .some(entry => entry.type === "service_tier_change"); const defaultThinkingLevel = parseConfiguredThinkingLevel(this.settings.get("defaultThinkingLevel")); - const configuredServiceTier = this.settings.get("serviceTier"); + const configuredServiceTierByFamily = buildServiceTierByFamily( + this.settings.get("tier.openai"), + this.settings.get("tier.anthropic"), + this.settings.get("tier.google"), + ); // Restore the thinking selector. Each change persists the configured // selector (`auto` or a concrete level), so prefer it: an `auto` session // resumes in auto mode (reclassifying the next turn) instead of freezing at @@ -13351,11 +13385,9 @@ export class AgentSession { this.#thinkingLevel = resolveThinkingLevelForModel(this.model, restoredThinkingLevel); } this.#applyThinkingLevelToAgent(this.#thinkingLevel); - this.agent.serviceTier = hasServiceTierEntry - ? sessionContext.serviceTier - : configuredServiceTier === "none" - ? undefined - : configuredServiceTier; + this.#serviceTierByFamily = hasServiceTierEntry + ? (sessionContext.serviceTier ?? {}) + : configuredServiceTierByFamily; if (switchingToDifferentSession) { await this.#resetMemoryContextForNewTranscript(); @@ -13412,7 +13444,7 @@ export class AgentSession { this.#autoThinking = previousAutoThinking; this.#autoResolvedLevel = previousAutoResolvedLevel; this.#applyThinkingLevelToAgent(previousThinkingLevel); - this.agent.serviceTier = previousServiceTier; + this.#serviceTierByFamily = previousServiceTierByFamily; this.#syncTodoPhasesFromBranch(); this.#resetAllAdvisorRuntimes(); this.#reconnectToAgent(); @@ -14366,7 +14398,7 @@ export class AgentSession { const payload = { model: this.agent.state.model ?? null, thinkingLevel: this.#thinkingLevel ?? null, - serviceTier: this.agent.serviceTier ?? null, + serviceTier: this.#serviceTierEntry(), systemPrompt: this.agent.state.systemPrompt, tools: this.agent.state.tools.map(tool => ({ name: tool.name, diff --git a/packages/coding-agent/src/session/session-context.ts b/packages/coding-agent/src/session/session-context.ts index 787da464a..79fe5ddea 100644 --- a/packages/coding-agent/src/session/session-context.ts +++ b/packages/coding-agent/src/session/session-context.ts @@ -1,5 +1,5 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; -import type { ProviderPayload, ServiceTier } from "@oh-my-pi/pi-ai"; +import { coerceServiceTierByFamily, type ProviderPayload, type ServiceTierByFamily } from "@oh-my-pi/pi-ai"; import * as snapcompact from "@oh-my-pi/snapcompact"; import { createBranchSummaryMessage, createCompactionSummaryMessage, createCustomMessage } from "./messages"; import { type CompactionEntry, EPHEMERAL_MODEL_CHANGE_ROLE, type SessionEntry } from "./session-entries"; @@ -9,7 +9,7 @@ export interface SessionContext { thinkingLevel?: string; /** Configured thinking selector (`"auto"` or a concrete level) from the latest change. */ configuredThinkingLevel?: string; - serviceTier?: ServiceTier; + serviceTier?: ServiceTierByFamily; /** Model roles: { default: "provider/modelId", small: "provider/modelId", ... } */ models: Record; /** Names of TTSR rules that have been injected this session */ @@ -138,7 +138,7 @@ export function buildSessionContext( // Extract settings and find compaction let thinkingLevel: string | undefined = "off"; let configuredThinkingLevel: string | undefined; - let serviceTier: ServiceTier | undefined; + let serviceTier: ServiceTierByFamily | undefined; const models: Record = {}; let compaction: CompactionEntry | null = null; const injectedTtsrRulesSet = new Set(); @@ -169,7 +169,7 @@ export function buildSessionContext( } } } else if (entry.type === "service_tier_change") { - serviceTier = entry.serviceTier ?? undefined; + serviceTier = coerceServiceTierByFamily(entry.serviceTier); } else if (entry.type === "message" && entry.message.role === "assistant") { // Legacy fallback: infer default model from assistant messages only // when no explicit `model_change` (role=default) entry has been diff --git a/packages/coding-agent/src/session/session-entries.ts b/packages/coding-agent/src/session/session-entries.ts index 324881ebf..830062697 100644 --- a/packages/coding-agent/src/session/session-entries.ts +++ b/packages/coding-agent/src/session/session-entries.ts @@ -1,5 +1,5 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; -import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai"; +import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai"; export const CURRENT_SESSION_VERSION = 3; @@ -73,7 +73,7 @@ export interface ModelChangeEntry extends SessionEntryBase { export interface ServiceTierChangeEntry extends SessionEntryBase { type: "service_tier_change"; - serviceTier: ServiceTier | null; + serviceTier: ServiceTierByFamily | null; } export interface CompactionEntry extends SessionEntryBase { diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index 8bb81ebfa..51a17ca1b 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -1,6 +1,13 @@ import * as fs from "node:fs"; import * as path from "node:path"; -import type { ImageContent, Message, MessageAttribution, ServiceTier, TextContent, Usage } from "@oh-my-pi/pi-ai"; +import type { + ImageContent, + Message, + MessageAttribution, + ServiceTierByFamily, + TextContent, + Usage, +} from "@oh-my-pi/pi-ai"; import { directoryExists, getBlobsDir, @@ -1286,7 +1293,7 @@ export class SessionManager { return entry.id; } - appendServiceTierChange(serviceTier: ServiceTier | null): string { + appendServiceTierChange(serviceTier: ServiceTierByFamily | null): string { const entry: ServiceTierChangeEntry = { type: "service_tier_change", ...this.#freshEntryFields(), serviceTier }; this.#recordEntry(entry); return entry.id; diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 91a6ead79..189967edc 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -73,17 +73,9 @@ function refreshStatusLine(ctx: InteractiveModeContext): void { ctx.ui.requestRender(); } -/** `/fast status` label: "off", "on", or scope-qualified "on (… only)". */ +/** `/fast status` label for the active model: "on" when its family is priority, else "off". */ function formatFastModeStatus(session: AgentSession): string { - if (!session.isFastModeEnabled()) return "off"; - switch (session.serviceTier) { - case "openai-only": - return "on (OpenAI only)"; - case "claude-only": - return "on (Claude only)"; - default: - return "on"; - } + return session.isFastModeEnabled() ? "on" : "off"; } const AUTOCOMPLETE_DETAIL_LIMIT = 48; diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 43b70bc82..76bf4e034 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -7,7 +7,7 @@ import path from "node:path"; import type { AgentEvent, AgentIdentity, AgentTelemetryConfig, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { recordHandoff, resolveTelemetry } from "@oh-my-pi/pi-agent-core"; -import type { Api, Model, ServiceTier, Usage } from "@oh-my-pi/pi-ai"; +import type { Api, Model, ServiceTierByFamily, Usage } from "@oh-my-pi/pi-ai"; import { logger, popLoopPhase, prompt, pushLoopPhase, untilAborted } from "@oh-my-pi/pi-utils"; import type { Rule } from "../capability/rule"; import { ModelRegistry } from "../config/model-registry"; @@ -18,7 +18,7 @@ import { resolveModelOverrideWithAuthFallback, } from "../config/model-resolver"; import type { PromptTemplate } from "../config/prompt-templates"; -import { resolveSubagentServiceTier } from "../config/service-tier"; +import { buildServiceTierByFamily, resolveSubagentServiceTier } from "../config/service-tier"; import { Settings } from "../config/settings"; import { SETTINGS_SCHEMA, type SettingPath } from "../config/settings-schema"; import type { ToolPathWithSource } from "../extensibility/custom-tools"; @@ -344,12 +344,12 @@ export interface ExecutorOptions { modelRegistry?: ModelRegistry; settings?: Settings; /** - * Parent session's live effective service tier, the source of truth for a - * subagent whose `serviceTierSubagent` is `"inherit"`. `null` = the parent + * Parent session's live per-family service tiers, the source of truth for a + * subagent whose `tier.subagent` is `"inherit"`. `null` = the parent * explicitly has no tier (e.g. `/fast off`); omitted = no live session, so - * inherit falls back to the configured `serviceTier` setting. + * inherit falls back to the subagent's configured `tier.*` settings. */ - parentServiceTier?: ServiceTier | null; + parentServiceTier?: ServiceTierByFamily | null; /** Override local:// protocol options so subagent shares parent's local:// root */ localProtocolOptions?: LocalProtocolOptions; /** @@ -739,21 +739,28 @@ export function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] { export function createSubagentSettings( baseSettings: Settings, overrides?: Partial>, - inheritedServiceTier?: ServiceTier | null, + inheritedServiceTier?: ServiceTierByFamily | null, ): Settings { const snapshot: Partial> = {}; for (const key of Object.keys(SETTINGS_SCHEMA) as SettingPath[]) { snapshot[key] = baseSettings.get(key); } - // Resolve the subagent's service tier from `serviceTierSubagent` ("inherit" = - // match the parent's live tier when a live session supplied one, else the - // configured `serviceTier`). The result is stamped back onto the snapshot so - // createAgentSession's `settings.get("serviceTier")` read picks it up. - snapshot.serviceTier = resolveSubagentServiceTier( - baseSettings.get("serviceTierSubagent"), - baseSettings.get("serviceTier"), - inheritedServiceTier, - ); + // Resolve the subagent's per-family tiers from `tier.subagent` ("inherit" = + // match the parent's live tiers when a live session supplied them, else the + // subagent's own configured tier.* settings). The result is stamped back onto + // the snapshot so createAgentSession's tier.* reads pick it up. + const inheritedTiers = + inheritedServiceTier === undefined + ? buildServiceTierByFamily( + baseSettings.get("tier.openai"), + baseSettings.get("tier.anthropic"), + baseSettings.get("tier.google"), + ) + : (inheritedServiceTier ?? {}); + const subagentTiers = resolveSubagentServiceTier(baseSettings.get("tier.subagent"), inheritedTiers); + snapshot["tier.openai"] = subagentTiers.openai ?? "none"; + snapshot["tier.anthropic"] = subagentTiers.anthropic ?? "none"; + snapshot["tier.google"] = subagentTiers.google ?? "none"; return Settings.isolated({ ...snapshot, "async.enabled": false, diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 0b235391d..d61208f65 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -1296,11 +1296,13 @@ export class TaskTool implements AgentTool => { diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 29e17fd0c..0be8c7d93 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -1,6 +1,6 @@ import type { InMemorySnapshotStore } from "@oh-my-pi/hashline"; import type { AgentTelemetryConfig, AgentTool } from "@oh-my-pi/pi-agent-core"; -import type { FetchImpl, ImageContent, Model, ServiceTier, ToolChoice } from "@oh-my-pi/pi-ai"; +import type { FetchImpl, ImageContent, Model, ServiceTierByFamily, ToolChoice } from "@oh-my-pi/pi-ai"; import { logger } from "@oh-my-pi/pi-utils"; import type { AsyncJobManager } from "../async/job-manager"; import type { Rule } from "../capability/rule"; @@ -240,8 +240,8 @@ export interface ToolSession { getActiveModelString?: () => string | undefined; /** Get the current session model object (provider/api capabilities), regardless of how it was chosen. */ getActiveModel?: () => Model | undefined; - /** Get the session's live effective service tier (undefined = none). Source of truth for subagent `serviceTierSubagent: inherit`. */ - getServiceTier?: () => ServiceTier | undefined; + /** Get the session's live per-family service tiers (undefined = none). Source of truth for subagent `tier.subagent: inherit`. */ + getServiceTierByFamily?: () => ServiceTierByFamily | undefined; /** Auth storage for passing to subagents (avoids re-discovery) */ authStorage?: import("../session/auth-storage").AuthStorage; /** Model registry for passing to subagents (avoids re-discovery) */ diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index d8e82aa54..516a4f906 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -452,7 +452,11 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, + { + role: "user", + content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], + timestamp: 1, + }, ], turnPrefixMessages: [], recentMessages: [], @@ -479,7 +483,11 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, + { + role: "user", + content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], + timestamp: 1, + }, ], turnPrefixMessages: [], recentMessages: [], diff --git a/packages/coding-agent/test/agent-session-mcp-discovery.test.ts b/packages/coding-agent/test/agent-session-mcp-discovery.test.ts index 36c44fffd..b642b72df 100644 --- a/packages/coding-agent/test/agent-session-mcp-discovery.test.ts +++ b/packages/coding-agent/test/agent-session-mcp-discovery.test.ts @@ -774,7 +774,7 @@ describe("AgentSession MCP discovery", () => { settings: Settings.isolated({ "mcp.discoveryMode": true, defaultThinkingLevel: "high", - serviceTier: "priority", + "tier.openai": "priority", }), modelRegistry: {} as never, toolRegistry, @@ -789,10 +789,10 @@ describe("AgentSession MCP discovery", () => { expect(session.getSelectedMCPToolNames()).toEqual(["mcp__docs_search"]); sessionManager.appendThinkingLevelChange(ThinkingLevel.High); - sessionManager.appendServiceTierChange("flex"); + sessionManager.appendServiceTierChange({ openai: "flex" }); sessionManager.appendMCPToolSelection(["mcp__docs_search"]); expect(sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.High); - expect(sessionManager.buildSessionContext().serviceTier).toBe("flex"); + expect(sessionManager.buildSessionContext().serviceTier).toEqual({ openai: "flex" }); expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual(["mcp__docs_search"]); expect(sessionManager.buildSessionContext().hasPersistedMCPToolSelection).toBe(true); await sessionManager.rewriteEntries(); @@ -803,7 +803,7 @@ describe("AgentSession MCP discovery", () => { await session.switchSession(olderSessionFile!); expect(session.sessionFile).toBe(olderSessionFile); expect(session.thinkingLevel).toBe(ThinkingLevel.Medium); - expect(session.serviceTier).toBe("priority"); + expect(session.serviceTierByFamily).toEqual({ openai: "priority" }); expect(session.getSelectedMCPToolNames()).toEqual([]); expect(session.getActiveToolNames()).toEqual(["read"]); expect(session.systemPrompt).toEqual(["tools:read"]); @@ -813,7 +813,7 @@ describe("AgentSession MCP discovery", () => { await session.switchSession(originalSessionFile!); expect(session.sessionFile).toBe(originalSessionFile); expect(session.thinkingLevel).toBe(ThinkingLevel.Medium); - expect(session.serviceTier).toBe("flex"); + expect(session.serviceTierByFamily).toEqual({ openai: "flex" }); expect(session.getSelectedMCPToolNames()).toEqual(["mcp__docs_search"]); expect(session.getActiveToolNames()).toEqual(["read", "mcp__docs_search"]); expect(session.systemPrompt).toEqual(["tools:read,mcp__docs_search"]); diff --git a/packages/coding-agent/test/bench-auth-fallback.test.ts b/packages/coding-agent/test/bench-auth-fallback.test.ts index aab7b4da0..219dda515 100644 --- a/packages/coding-agent/test/bench-auth-fallback.test.ts +++ b/packages/coding-agent/test/bench-auth-fallback.test.ts @@ -173,13 +173,16 @@ describe("bench empty-output guard", () => { function settingsStub(serviceTier: string | undefined): Settings | undefined { if (serviceTier === undefined) return undefined; - return { get: (key: string) => (key === "serviceTier" ? serviceTier : undefined) } as unknown as Settings; + return { + get: (key: string) => + key === "tier.openai" ? serviceTier : key === "tier.anthropic" || key === "tier.google" ? "none" : undefined, + } as unknown as Settings; } async function captureServiceTier(opts: { flag?: string; setting?: string; -}): Promise<{ wire: SimpleStreamOptions["serviceTier"]; summary: BenchSummary["serviceTier"] }> { +}): Promise<{ wire: SimpleStreamOptions["serviceTier"]; summary: BenchSummary["serviceTierByFamily"] }> { const registry = fakeRegistry({ models: [fakeModel("openai-codex", "gpt-5.5")], authedProviders: ["openai-codex"] }); let captured: SimpleStreamOptions | undefined; const summary = await runBenchCommand( @@ -205,7 +208,7 @@ async function captureServiceTier(opts: { stdoutIsTTY: false, }, ); - return { wire: captured?.serviceTier, summary: summary.serviceTier }; + return { wire: captured?.serviceTier, summary: summary.serviceTierByFamily }; } describe("bench provider session state and websocket preference", () => { @@ -245,24 +248,24 @@ describe("bench service tier", () => { it("sends the configured serviceTier setting when no flag is passed", async () => { const { wire, summary } = await captureServiceTier({ setting: "flex" }); expect(wire).toBe("flex"); - expect(summary).toBe("flex"); + expect(summary).toEqual({ openai: "flex" }); }); it("lets an explicit --service-tier override the configured setting", async () => { const { wire, summary } = await captureServiceTier({ flag: "priority", setting: "flex" }); expect(wire).toBe("priority"); - expect(summary).toBe("priority"); + expect(summary).toEqual({ openai: "priority", anthropic: "priority", google: "priority" }); }); it("omits service_tier when the setting is none and no flag is passed", async () => { const { wire, summary } = await captureServiceTier({ setting: "none" }); expect(wire).toBeUndefined(); - expect(summary).toBeUndefined(); + expect(summary).toEqual({}); }); it("omits service_tier when neither flag nor settings are present", async () => { const { wire, summary } = await captureServiceTier({}); expect(wire).toBeUndefined(); - expect(summary).toBeUndefined(); + expect(summary).toEqual({}); }); }); diff --git a/packages/coding-agent/test/fast-mode-scope.test.ts b/packages/coding-agent/test/fast-mode-scope.test.ts index 9a92d7a05..10f70fb43 100644 --- a/packages/coding-agent/test/fast-mode-scope.test.ts +++ b/packages/coding-agent/test/fast-mode-scope.test.ts @@ -9,9 +9,7 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { TempDir } from "@oh-my-pi/pi-utils"; -type FastModeScope = "both" | "openai" | "claude"; - -describe("fast mode scope", () => { +describe("/fast targets the current model's service-tier family", () => { let tempDir: TempDir; let authStorage: AuthStorage; let session: AgentSession; @@ -29,77 +27,54 @@ describe("fast mode scope", () => { tempDir.removeSync(); }); - async function createSession(fastModeScope?: FastModeScope): Promise { - const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + async function createSession(provider: "anthropic" | "openai", modelId: string): Promise { + const model = getBundledModel(provider, modelId); if (!model) { - throw new Error("Expected bundled test model to exist"); + throw new Error(`Expected bundled test model ${provider}/${modelId} to exist`); } - - const settings = fastModeScope === undefined ? Settings.isolated() : Settings.isolated({ fastModeScope }); const agent = new Agent({ - initialState: { - model, - systemPrompt: ["Test"], - tools: [], - messages: [], - }, + initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] }, }); - authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); - authStorage.setRuntimeApiKey(model.provider, "anthropic-token"); + authStorage.setRuntimeApiKey(model.provider, "token"); modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - session = new AgentSession({ agent, sessionManager: SessionManager.inMemory(), - settings, + settings: Settings.isolated(), modelRegistry, }); session.subscribe(() => {}); return session; } - it("scopes enabled fast mode to OpenAI when configured", async () => { - const session = await createSession("openai"); - + it("enables priority on the Anthropic family for a Claude model", async () => { + const session = await createSession("anthropic", "claude-sonnet-4-5"); session.setFastMode(true); - - expect(session.serviceTier).toBe("openai-only"); + expect(session.serviceTierByFamily).toEqual({ anthropic: "priority" }); + expect(session.isFastModeEnabled()).toBe(true); }); - it("scopes enabled fast mode to Claude when configured", async () => { - const session = await createSession("claude"); - + it("enables priority on the OpenAI family for an OpenAI model", async () => { + const session = await createSession("openai", "gpt-5.2"); session.setFastMode(true); - - expect(session.serviceTier).toBe("claude-only"); + expect(session.serviceTierByFamily).toEqual({ openai: "priority" }); + expect(session.isFastModeEnabled()).toBe(true); }); - it("defaults enabled fast mode to priority for both providers", async () => { - const session = await createSession(); - + it("clears only the current model's family when disabled", async () => { + const session = await createSession("anthropic", "claude-sonnet-4-5"); session.setFastMode(true); - - expect(session.serviceTier).toBe("priority"); - }); - - it("clears the service tier when disabled", async () => { - const session = await createSession("openai"); - session.setFastMode(true); - session.setFastMode(false); - - expect(session.serviceTier).toBeUndefined(); + expect(session.serviceTierByFamily).toEqual({}); + expect(session.isFastModeEnabled()).toBe(false); }); - it("does not broaden an already enabled scoped tier", async () => { - const session = await createSession("claude"); - session.setFastMode(true); - expect(session.serviceTier).toBe("claude-only"); - session.settings.set("fastModeScope", "both"); - - session.setFastMode(true); - - expect(session.serviceTier).toBe("claude-only"); + it("toggle reports the resulting state", async () => { + const session = await createSession("anthropic", "claude-sonnet-4-5"); + expect(session.toggleFastMode()).toBe(true); + expect(session.serviceTierByFamily.anthropic).toBe("priority"); + expect(session.toggleFastMode()).toBe(false); + expect(session.serviceTierByFamily.anthropic).toBeUndefined(); }); }); diff --git a/packages/coding-agent/test/sdk-mcp-discovery.test.ts b/packages/coding-agent/test/sdk-mcp-discovery.test.ts index f4036ae67..71deb1a66 100644 --- a/packages/coding-agent/test/sdk-mcp-discovery.test.ts +++ b/packages/coding-agent/test/sdk-mcp-discovery.test.ts @@ -331,7 +331,7 @@ describe("createAgentSession MCP discovery prompt gating", () => { settings: Settings.isolated({ "mcp.discoveryMode": true, defaultThinkingLevel: "high", - serviceTier: "priority", + "tier.openai": "priority", }), model: createReasoningModel(), disableExtensionDiscovery: true, @@ -349,7 +349,7 @@ describe("createAgentSession MCP discovery prompt gating", () => { }); await firstSession.activateDiscoveredMCPTools(["mcp__slack_post_message"]); firstSession.sessionManager.appendThinkingLevelChange(ThinkingLevel.Off); - firstSession.sessionManager.appendServiceTierChange("priority"); + firstSession.sessionManager.appendServiceTierChange({ openai: "priority" }); expect(firstSession.sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.Off); expect(firstSession.getSelectedMCPToolNames()).toEqual(["mcp__slack_post_message"]); const sessionFile = firstSession.sessionFile; @@ -368,7 +368,7 @@ describe("createAgentSession MCP discovery prompt gating", () => { settings: Settings.isolated({ "mcp.discoveryMode": true, defaultThinkingLevel: "high", - serviceTier: "none", + "tier.openai": "none", }), model: createReasoningModel(), disableExtensionDiscovery: true, @@ -386,7 +386,7 @@ describe("createAgentSession MCP discovery prompt gating", () => { }); try { expect(resumedSession.thinkingLevel).toBe(ThinkingLevel.Off); - expect(resumedSession.serviceTier).toBe("priority"); + expect(resumedSession.serviceTierByFamily).toEqual({ openai: "priority" }); expect(resumedSession.getSelectedMCPToolNames()).toEqual(["mcp__slack_post_message"]); expect(resumedSession.getActiveToolNames()).toEqual( expect.arrayContaining(["read", "search_tool_bm25", "mcp__slack_post_message"]), @@ -422,7 +422,7 @@ describe("createAgentSession MCP discovery prompt gating", () => { "mcp.discoveryMode": true, "mcp.discoveryDefaultServers": ["github"], defaultThinkingLevel: "high", - serviceTier: "priority", + "tier.openai": "priority", }), model: createReasoningModel(), disableExtensionDiscovery: true, @@ -440,7 +440,7 @@ describe("createAgentSession MCP discovery prompt gating", () => { }); try { expect(session.thinkingLevel).toBe(ThinkingLevel.High); - expect(session.serviceTier).toBe("priority"); + expect(session.serviceTierByFamily).toEqual({ openai: "priority" }); expect(session.getSelectedMCPToolNames()).toEqual(["mcp__github_create_issue"]); expect(session.getActiveToolNames()).toEqual( expect.arrayContaining(["read", "search_tool_bm25", "mcp__github_create_issue"]), diff --git a/packages/coding-agent/test/service-tier-migration.test.ts b/packages/coding-agent/test/service-tier-migration.test.ts new file mode 100644 index 000000000..392f34320 --- /dev/null +++ b/packages/coding-agent/test/service-tier-migration.test.ts @@ -0,0 +1,82 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as path from "node:path"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage"; +import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils"; +import { YAML } from "bun"; +import { beginSettingsTest, restoreSettingsTestState, type SettingsTestState } from "./helpers/settings-test-state"; + +// Locks the back-compat migration of the legacy single `serviceTier` enum (with +// scoped `openai-only`/`claude-only` sentinels) plus `serviceTierSubagent`/ +// `serviceTierAdvisor`/`fastModeScope` into the per-family `tier.*` settings. +describe("serviceTier → tier.* settings migration", () => { + let settingsState: SettingsTestState | undefined; + let tempDir: TempDir; + let agentDir: string; + let projectDir: string; + + beforeEach(() => { + settingsState = beginSettingsTest(); + tempDir = TempDir.createSync("@test-service-tier-migration-"); + agentDir = path.join(tempDir.path(), "agent"); + projectDir = path.join(tempDir.path(), "project"); + fs.mkdirSync(agentDir, { recursive: true }); + fs.mkdirSync(getProjectAgentDir(projectDir), { recursive: true }); + }); + + afterEach(async () => { + AgentStorage.resetInstance(); + restoreSettingsTestState(settingsState); + settingsState = undefined; + try { + await tempDir.remove(); + } catch {} + }); + + async function loadWith(raw: Record): Promise { + await Bun.write(path.join(agentDir, "config.yml"), YAML.stringify(raw, null, 2)); + resetSettingsForTest(); + return Settings.init({ cwd: projectDir, agentDir }); + } + + it("expands unscoped priority to every family", async () => { + const settings = await loadWith({ serviceTier: "priority" }); + expect(settings.get("tier.openai")).toBe("priority"); + expect(settings.get("tier.anthropic")).toBe("priority"); + expect(settings.get("tier.google")).toBe("priority"); + }); + + it("scopes openai-only/claude-only to a single family", async () => { + const openai = await loadWith({ serviceTier: "openai-only" }); + expect(openai.get("tier.openai")).toBe("priority"); + expect(openai.get("tier.anthropic")).toBe("none"); + expect(openai.get("tier.google")).toBe("none"); + + const claude = await loadWith({ serviceTier: "claude-only" }); + expect(claude.get("tier.anthropic")).toBe("priority"); + expect(claude.get("tier.openai")).toBe("none"); + }); + + it("maps plain OpenAI tiers onto the OpenAI family", async () => { + const settings = await loadWith({ serviceTier: "flex" }); + expect(settings.get("tier.openai")).toBe("flex"); + expect(settings.get("tier.anthropic")).toBe("none"); + }); + + it("carries subagent/advisor over and drops scoped sentinels", async () => { + const settings = await loadWith({ + serviceTierSubagent: "claude-only", + serviceTierAdvisor: "flex", + }); + expect(settings.get("tier.subagent")).toBe("priority"); // claude-only → priority + expect(settings.get("tier.advisor")).toBe("flex"); + }); + + it("leaves a fresh config on the per-family defaults", async () => { + const settings = await loadWith({}); + expect(settings.get("tier.openai")).toBe("none"); + expect(settings.get("tier.subagent")).toBe("inherit"); + expect(settings.get("tier.advisor")).toBe("none"); + }); +}); diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index 81c77b1cc..fcfe5d6dc 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Improved premium request calculation logic to account for specific model families + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/stats/src/parser.ts b/packages/stats/src/parser.ts index d8453a930..b77f9e549 100644 --- a/packages/stats/src/parser.ts +++ b/packages/stats/src/parser.ts @@ -1,6 +1,12 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; -import { type AssistantMessage, getPriorityPremiumRequests, type ServiceTier } from "@oh-my-pi/pi-ai"; +import { + type AssistantMessage, + coerceServiceTierByFamily, + getPriorityPremiumRequests, + resolveModelServiceTier, + type ServiceTierByFamily, +} from "@oh-my-pi/pi-ai"; import { getSessionsDir, isEnoent } from "@oh-my-pi/pi-utils"; import type { AgentType, @@ -130,7 +136,7 @@ function extractStats( sessionFile: string, folder: string, entry: SessionMessageEntry, - currentServiceTier: ServiceTier | undefined, + currentServiceTier: ServiceTierByFamily | undefined, agentType: AgentType, ): MessageStats | null { const msg = entry.message as AssistantMessage; @@ -143,7 +149,9 @@ function extractStats( // non-zero value already in `usage.premiumRequests` (Copilot multipliers or // the new AI code path) and only synthesise when the field is missing/zero. const recorded = msg.usage.premiumRequests ?? 0; - const derived = recorded > 0 ? recorded : getPriorityPremiumRequests(currentServiceTier, msg.provider); + const model = { provider: msg.provider, api: msg.api, id: msg.model }; + const tier = resolveModelServiceTier(currentServiceTier, model); + const derived = recorded > 0 ? recorded : getPriorityPremiumRequests(tier, model); const usage = derived === recorded ? msg.usage : { ...msg.usage, premiumRequests: derived }; return { @@ -206,10 +214,10 @@ function parseSessionEntriesLenient(bytes: Uint8Array): { entries: SessionEntry[ return { entries, read }; } -function scanLastServiceTier(bytes: Uint8Array): ServiceTier | undefined { - let currentServiceTier: ServiceTier | undefined; +function scanLastServiceTier(bytes: Uint8Array): ServiceTierByFamily | undefined { + let currentServiceTier: ServiceTierByFamily | undefined; visitSessionEntriesLenient(bytes, entry => { - if (isServiceTierChange(entry)) currentServiceTier = entry.serviceTier ?? undefined; + if (isServiceTierChange(entry)) currentServiceTier = coerceServiceTierByFamily(entry.serviceTier); }); return currentServiceTier; } @@ -253,13 +261,13 @@ export async function parseSessionFile(sessionPath: string, fromOffset = 0): Pro const start = Math.max(0, Math.min(fromOffset, bytes.length)); const unprocessed = bytes.subarray(start); const { entries, read } = parseSessionEntriesLenient(unprocessed); - let currentServiceTier: ServiceTier | undefined; + let currentServiceTier: ServiceTierByFamily | undefined; if (start > 0) { currentServiceTier = scanLastServiceTier(bytes.subarray(0, start)); } for (const entry of entries) { if (isServiceTierChange(entry)) { - currentServiceTier = entry.serviceTier ?? undefined; + currentServiceTier = coerceServiceTierByFamily(entry.serviceTier); continue; } if (isUserMessage(entry)) { diff --git a/packages/stats/src/types.ts b/packages/stats/src/types.ts index 843e27e32..8036e5077 100644 --- a/packages/stats/src/types.ts +++ b/packages/stats/src/types.ts @@ -1,4 +1,4 @@ -import type { AssistantMessage, ServiceTier, StopReason, Usage } from "@oh-my-pi/pi-ai"; +import type { AssistantMessage, ServiceTier, ServiceTierByFamily, StopReason, Usage } from "@oh-my-pi/pi-ai"; import type { AgentType } from "./shared-types"; export * from "./shared-types"; @@ -72,7 +72,7 @@ export interface SessionServiceTierChangeEntry { id: string; parentId?: string | null; timestamp: string; - serviceTier: ServiceTier | null; + serviceTier: ServiceTierByFamily | ServiceTier | null; } export type SessionEntry = SessionHeader | SessionMessageEntry | SessionServiceTierChangeEntry | { type: string }; From fa2e3a807f48a35440b427373c0a0da757849934 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 02:13:17 +0000 Subject: [PATCH 26/88] style: bun run fix --- packages/ai/CHANGELOG.md | 4 ++ .../__tests__/kimi-code-thinking.test.ts | 55 ++++++++++++++++++- packages/ai/src/providers/anthropic.ts | 11 ++-- packages/catalog/CHANGELOG.md | 2 +- packages/catalog/src/compat/anthropic.ts | 11 +++- packages/catalog/src/compat/openai.ts | 2 +- packages/catalog/src/models.json | 1 + packages/catalog/src/types.ts | 5 ++ .../test/agent-session-handoff.test.ts | 12 +++- 9 files changed, 90 insertions(+), 13 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3a4f8c78d..4b1fc6925 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts index 0b3c08de4..4dcab4218 100644 --- a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts +++ b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from "bun:test"; import { getBundledModel } from "@oh-my-pi/pi-catalog"; +import type { Context } from "../../types"; +import type { MessageCreateParamsStreaming } from "../anthropic-wire"; +import { streamOpenAIAnthropicShim } from "../openai-anthropic-shim"; import { applyChatCompletionsCompatPolicy, type OpenAICompletionsParams, @@ -7,10 +10,26 @@ import { } from "../openai-shared"; const BASE_CHAT_COMPLETIONS_PARAMS: OpenAICompletionsParams = { messages: [], model: "unused", stream: true }; +const TITLE_CONTEXT: Context = { + systemPrompt: ["Generate a title."], + messages: [{ role: "user", content: "Explain the login failure", timestamp: 0 }], + tools: [ + { + name: "set_title", + description: "Set title", + parameters: { + type: "object", + properties: { title: { type: "string" } }, + required: ["title"], + additionalProperties: false, + }, + }, + ], +}; describe("Kimi K2.7 Code thinking policy", () => { it("omits disabled thinking for title-generator-style Kimi Code requests", () => { - const model = getBundledModel("kimi-code", "kimi-for-coding"); + const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); const policy = resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", disableReasoning: true, @@ -21,11 +40,40 @@ describe("Kimi K2.7 Code thinking policy", () => { applyChatCompletionsCompatPolicy(params, policy); expect("thinking" in params).toBe(false); + expect(model.compat.supportsForcedToolChoice).toBe(false); + }); + + it("enables thinking and downgrades forced tool choice on Kimi Code's Anthropic endpoint", async () => { + const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); + let payload: MessageCreateParamsStreaming | undefined; + const stream = streamOpenAIAnthropicShim( + model, + TITLE_CONTEXT, + { + apiKey: "test-key", + maxTokens: 1024, + disableReasoning: true, + toolChoice: { type: "tool", name: "set_title" }, + onPayload: body => { + payload = body as MessageCreateParamsStreaming; + throw new Error("stop after payload capture"); + }, + }, + { + anthropicBaseUrl: "https://api.kimi.com/coding", + defaultFormat: "anthropic", + }, + ); + + await stream.result(); + + expect(payload?.thinking?.type).toBe("enabled"); + expect(payload?.tool_choice).toEqual({ type: "auto" }); }); it("omits disabled thinking for native Moonshot Kimi K2.7 Code variants", () => { for (const modelId of ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"]) { - const model = getBundledModel("moonshot", modelId); + const model = getBundledModel<"openai-completions">("moonshot", modelId); const policy = resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", disableReasoning: true, @@ -34,11 +82,12 @@ describe("Kimi K2.7 Code thinking policy", () => { applyChatCompletionsCompatPolicy(params, policy); expect("thinking" in params).toBe(false); + expect(model.compat.supportsForcedToolChoice).toBe(false); } }); it("keeps explicit disabled thinking for Kimi K2.6", () => { - const model = getBundledModel("moonshot", "kimi-k2.6"); + const model = getBundledModel<"openai-completions">("moonshot", "kimi-k2.6"); const policy = resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", disableReasoning: true, diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 442cfd59e..4e9dbf81d 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2876,9 +2876,10 @@ function buildParams( let thinking: MessageCreateParamsStreaming["thinking"] | undefined; let outputConfigEffort: AnthropicOutputEffort | undefined; if (model.reasoning) { - if (options?.thinkingEnabled) { + if (options?.thinkingEnabled || model.compat.requiresThinkingEnabled) { + const thinkingOptions = options ?? {}; const mode = model.thinking?.mode; - const effort = resolveAnthropicAdaptiveEffort(model, options); + const effort = resolveAnthropicAdaptiveEffort(model, thinkingOptions); const compat = model.compat; if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" }; @@ -2889,15 +2890,15 @@ function buildParams( // support: Opus 4.6 / Sonnet 4.6+ reject it with a 400, so an explicit // `thinkingDisplay` MUST NOT force it onto a model that can't accept it. if (model.thinking?.supportsDisplay) { - adaptive.display = options.thinkingDisplay ?? "summarized"; + adaptive.display = thinkingOptions.thinkingDisplay ?? "summarized"; } thinking = adaptive; if (effort && effort !== "adaptive") outputConfigEffort = effort; } else { thinking = { type: "enabled", - budget_tokens: options.thinkingBudgetTokens || 1024, - display: options.thinkingDisplay ?? "summarized", + budget_tokens: thinkingOptions.thinkingBudgetTokens || 1024, + display: thinkingOptions.thinkingDisplay ?? "summarized", }; if (mode === "anthropic-budget-effort" && effort && effort !== "adaptive") outputConfigEffort = effort; } diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index dc60dc92a..4d2edc7d5 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Kimi K2.7 Code compatibility to avoid sending disabled thinking to native Kimi endpoints that require thinking mode. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) +- Fixed Kimi K2.7 Code compatibility to avoid disabled thinking and forced tool choice on native Kimi endpoints that require thinking mode. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) ## [16.2.6] - 2026-06-29 diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index 93390b0e3..876f11f3c 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -28,6 +28,13 @@ export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean { return lower === OFFICIAL_ANTHROPIC_URL || lower.startsWith(`${OFFICIAL_ANTHROPIC_URL}/`); } +const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i; + +function requiresKimiK27CodeEnabledThinking(spec: ModelSpec<"anthropic-messages">): boolean { + if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true; + return spec.provider === "kimi-code" && spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); +} + /** Build the resolved anthropic-messages compat record for a model spec. */ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): ResolvedAnthropicCompat { const baseUrl = spec.baseUrl; @@ -40,6 +47,7 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // doesn't whitelist the `fine-grained-tool-streaming-2025-05-14` beta either // (issue #2558), so eager tool-input streaming is unavailable on this host. const isCopilot = modelMatchesHost(spec, "githubCopilot"); + const requiresThinkingEnabled = requiresKimiK27CodeEnabledThinking(spec); const compat: ResolvedAnthropicCompat = { officialEndpoint: official, disableStrictTools: false, @@ -53,12 +61,13 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // detection requires the canonical api.anthropic.com host plus a // supported model id. supportsMidConversationSystem: official && supportsMidConversationSystemMessages(spec.id), - supportsForcedToolChoice: !isAnthropicFableOrMythosModel(spec.id), + supportsForcedToolChoice: !requiresThinkingEnabled && !isAnthropicFableOrMythosModel(spec.id), // Opus 4.7+ and Fable/Mythos reject temperature/top_p/top_k with a 400. supportsSamplingParams: !hasOpus47ApiRestrictions(spec.id), // Z.AI workaround (issue #814): its proxy deserializes tool_result blocks // into a class that reads `.id`. requiresToolResultId: isZai, + requiresThinkingEnabled, // Official Anthropic enforces signature-based thinking-chain integrity, so // unsigned thinking blocks must stay text there. Anthropic-compatible // reasoning endpoints commonly emit unsigned thinking blocks while still diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 42ab2f305..c3ae7e2d6 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -411,7 +411,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel, disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter, supportsToolChoice: !isDirectDeepseekReasoning, - supportsForcedToolChoice: true, + supportsForcedToolChoice: !requiresEnabledThinking, supportsNamedToolChoice: provider !== "llama.cpp", maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens", requiresToolResultName: isMistral, diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 877830895..a95d863af 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -75779,6 +75779,7 @@ "supportsForcedToolChoice": true, "supportsSamplingParams": true, "requiresToolResultId": false, + "requiresThinkingEnabled": false, "replayUnsignedThinking": false, "escapeBuiltinToolNames": false } diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 29c0a9571..c41d76d3c 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -420,6 +420,11 @@ export interface AnthropicCompat { * Default: auto-detected from provider/baseUrl and `model.reasoning`. */ replayUnsignedThinking?: boolean; + /** + * Whether the endpoint requires `thinking.type: "enabled"` whenever the + * model reasons. Use for models that reject omitted or disabled thinking. + */ + requiresThinkingEnabled?: boolean; /** * Prefix Anthropic built-in tool names (`web_search`, `code_execution`, ...) * when they are ordinary client tools. Some Anthropic-compatible gateways diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index d8e82aa54..516a4f906 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -452,7 +452,11 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, + { + role: "user", + content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], + timestamp: 1, + }, ], turnPrefixMessages: [], recentMessages: [], @@ -479,7 +483,11 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, + { + role: "user", + content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], + timestamp: 1, + }, ], turnPrefixMessages: [], recentMessages: [], From 7cab14d963941a45fca3065c9c04d5cd34f29c68 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 02:30:40 +0000 Subject: [PATCH 27/88] fix(catalog): gated kimi K2.7 Code thinking-required policy on native Moonshot/Kimi-code host Reviewer flagged that the omit/forced-tool gate matched every public id, regressing Fireworks and OpenRouter Kimi K2.7 Code (non-zai dialects). Added Fireworks + OpenRouter regression coverage. --- .../__tests__/kimi-code-thinking.test.ts | 11 +++++++++++ packages/catalog/src/compat/anthropic.ts | 7 ++++--- packages/catalog/src/compat/openai.ts | 15 +++++++++++---- 3 files changed, 26 insertions(+), 7 deletions(-) diff --git a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts index 4dcab4218..b76cbd6cb 100644 --- a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts +++ b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts @@ -86,6 +86,17 @@ describe("Kimi K2.7 Code thinking policy", () => { } }); + it("keeps the openai disable shape for non-native Kimi K2.7 Code aliases", () => { + for (const { provider, id } of [ + { provider: "fireworks", id: "kimi-k2.7-code" }, + { provider: "openrouter", id: "moonshotai/kimi-k2.7-code" }, + ] as const) { + const model = getBundledModel<"openai-completions">(provider, id); + expect(model.compat.supportsForcedToolChoice).toBe(true); + expect(model.compat.reasoningDisableMode).not.toBe("omit"); + } + }); + it("keeps explicit disabled thinking for Kimi K2.6", () => { const model = getBundledModel<"openai-completions">("moonshot", "kimi-k2.6"); const policy = resolveOpenAICompatPolicy(model, { diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index 876f11f3c..cdff05ec2 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -28,11 +28,12 @@ export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean { return lower === OFFICIAL_ANTHROPIC_URL || lower.startsWith(`${OFFICIAL_ANTHROPIC_URL}/`); } +/** Mirrors `compat/openai.ts`; native-only host gating is the caller's responsibility. */ const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i; -function requiresKimiK27CodeEnabledThinking(spec: ModelSpec<"anthropic-messages">): boolean { +function matchesKimiK27CodeFamily(spec: ModelSpec<"anthropic-messages">): boolean { if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true; - return spec.provider === "kimi-code" && spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); + return spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); } /** Build the resolved anthropic-messages compat record for a model spec. */ @@ -47,7 +48,7 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // doesn't whitelist the `fine-grained-tool-streaming-2025-05-14` beta either // (issue #2558), so eager tool-input streaming is unavailable on this host. const isCopilot = modelMatchesHost(spec, "githubCopilot"); - const requiresThinkingEnabled = requiresKimiK27CodeEnabledThinking(spec); + const requiresThinkingEnabled = modelMatchesHost(spec, "moonshotNative") && matchesKimiK27CodeFamily(spec); const compat: ResolvedAnthropicCompat = { officialEndpoint: official, disableStrictTools: false, diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index c3ae7e2d6..d0ca9447e 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -39,12 +39,19 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; /** Kimi K2.6 can spend several minutes reasoning before the first visible token. */ const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; -/** Native Kimi K2.7 Code requires thinking; sending `thinking: { type: "disabled" }` 400s. */ +/** + * Native Kimi K2.7 Code requires `thinking.type: "enabled"` and rejects + * disabled thinking. Match the public id, its Fast variant, and the + * `kimi-code/kimi-for-coding` alias (which keeps the family name). + * Caller-disabled requests on non-native dialects (Fireworks `openai`, + * OpenRouter `openrouter`, …) MUST keep their per-dialect disable shape — + * gating on `isMoonshotKimi` is the caller's responsibility. + */ const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i; -function requiresKimiK27CodeEnabledThinking(spec: ModelSpec<"openai-completions">): boolean { +function matchesKimiK27CodeFamily(spec: ModelSpec<"openai-completions">): boolean { if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true; - return spec.provider === "kimi-code" && spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); + return spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); } /** Xiaomi MiMo Pro on api.xiaomimimo.com can stall ~2min before the first event (issue #1770). */ const XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS = 300_000; @@ -238,7 +245,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const isKimiModel = isKimiModelId(spec.id); const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative"); const isMoonshotKimi = isKimiModel && isMoonshotNative; - const requiresEnabledThinking = requiresKimiK27CodeEnabledThinking(spec); + const requiresEnabledThinking = isMoonshotKimi && matchesKimiK27CodeFamily(spec); const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id); const isAnthropicModel = modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id); From 6b64ed0ff6dce76c55d5eff44ce0ae348ce937dc Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 04:48:01 +0200 Subject: [PATCH 28/88] feat(ai/providers): added global fallback for regional vertex endpoints - Introduce `fallbackUrl` to allow retrying requests on the global Vertex endpoint when a 404 is encountered on a regional endpoint. - Correctly track tool call names to facilitate accurate function response mapping. - Update `serviceTier` type definitions to accurately reflect allowed values. - Refine vertex endpoint location resolution to ensure better compatibility with ambient regional settings. --- packages/ai/CHANGELOG.md | 4 +- packages/ai/src/providers/google-shared.ts | 61 +++++++++++++++++----- packages/ai/src/providers/google-types.ts | 5 +- packages/ai/src/providers/google-vertex.ts | 22 ++++++-- 4 files changed, 69 insertions(+), 23 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3de8e047d..5e64524fc 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -15,11 +15,11 @@ ### Fixed +- Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors + - Fixed safety setting application for Google Vertex AI models - Ensured Gemini service tier is correctly passed through to the API - Corrected priority request accounting for supported providers -### Fixed - - Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) ## [16.2.6] - 2026-06-29 diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index f2fe100d8..ae5cd9d0b 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -130,11 +130,9 @@ function resolveThoughtSignature(isSameProviderAndModel: boolean, signature: str return isSameProviderAndModel && isValidThoughtSignature(signature) ? signature : undefined; } -/** - * Claude models via Google APIs require explicit tool call IDs in function calls/responses. - */ -export function requiresToolCallId(modelId: string): boolean { - return modelId.startsWith("claude-"); +function supportsFunctionPartId(model: Model): boolean { + if (model.api === "google-vertex") return false; + return model.id.startsWith("claude-") || (model.api === "google-generative-ai" && isGemini3Model(model.id)); } function getGeminiMajorVersion(modelId: string): number | undefined { @@ -160,8 +158,9 @@ function isGemini3Model(modelId: string): boolean { */ export function convertMessages(model: Model, context: Context): Content[] { const contents: Content[] = []; + const emittedToolCallNames = new Map(); + const normalizeToolCallId = (id: string): string => { - if (!requiresToolCallId(model.id)) return id; return id.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64); }; @@ -247,6 +246,7 @@ export function convertMessages(model: Model, contex }); } } else if (block.type === "toolCall") { + emittedToolCallNames.set(block.id, block.name); const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thoughtSignature); const effectiveSignature = thoughtSignature || (isGemini3Model(model.id) ? SKIP_THOUGHT_SIGNATURE : undefined); @@ -255,11 +255,11 @@ export function convertMessages(model: Model, contex functionCall: { name: block.name, args: block.arguments ?? {}, - ...(requiresToolCallId(model.id) ? { id: block.id } : {}), + ...(supportsFunctionPartId(model) ? { id: block.id } : {}), }, }; if (model.provider === "google-vertex" && part?.functionCall?.id) { - delete part.functionCall.id; // Vertex AI does not support 'id' in functionCall + delete part.functionCall.id; // Vertex AI GenerateContent rejects 'id' in functionCall parts. } if (effectiveSignature) { part.thoughtSignature = effectiveSignature; @@ -305,10 +305,11 @@ export function convertMessages(model: Model, contex }, })); - const includeId = requiresToolCallId(model.id); + const includeId = supportsFunctionPartId(model); + const emittedName = emittedToolCallNames.get(msg.toolCallId); const functionResponsePart: Part = { functionResponse: { - name: msg.toolName, + name: emittedName ?? msg.toolName, response: msg.isError ? { error: responseValue } : { output: responseValue }, ...(hasImages && modelSupportsMultimodalFunctionResponse && { parts: imageParts }), ...(includeId ? { id: msg.toolCallId } : {}), @@ -316,7 +317,7 @@ export function convertMessages(model: Model, contex }; if (model.provider === "google-vertex" && functionResponsePart.functionResponse?.id) { - delete functionResponsePart.functionResponse.id; // Vertex AI does not support 'id' in functionResponse + delete functionResponsePart.functionResponse.id; // Vertex AI GenerateContent rejects 'id' in functionResponse parts. } // Cloud Code Assist API requires all function responses to be in a single user turn. @@ -533,6 +534,24 @@ export function pushToolCallEvents( * `text_start` / `thinking_start` event. `onBeforeStartEvent` lets the SSE consumer * inject its `ensureStarted()` first-token side effect into the canonical event order. */ +export function startTextOrThinkingBlock( + isThinking: true, + output: AssistantMessage, + stream: AssistantMessageEventStream, + onBeforeStartEvent?: () => void, +): ThinkingContent; +export function startTextOrThinkingBlock( + isThinking: false, + output: AssistantMessage, + stream: AssistantMessageEventStream, + onBeforeStartEvent?: () => void, +): TextContent; +export function startTextOrThinkingBlock( + isThinking: boolean, + output: AssistantMessage, + stream: AssistantMessageEventStream, + onBeforeStartEvent?: () => void, +): TextContent | ThinkingContent; export function startTextOrThinkingBlock( isThinking: boolean, output: AssistantMessage, @@ -864,6 +883,8 @@ export interface GoogleGenAIRequestPlan { url: string; headers: Record; fetch?: FetchImpl; + /** Optional URL retried once when {@link url} returns 404 (regional Vertex endpoint missing a global-only model). */ + fallbackUrl?: string; } export function streamGoogleGenAI(args: { @@ -918,8 +939,8 @@ export function streamGoogleGenAI> => { - const response = await fetchImpl(plan.url, { + const openStreamAt = async (requestUrl: string): Promise> => { + const response = await fetchImpl(requestUrl, { method: "POST", headers: { ...plan.headers, "Content-Type": "application/json", Accept: "text/event-stream" }, body: bodyJson, @@ -941,6 +962,20 @@ export function streamGoogleGenAI; }; + // A regional Vertex endpoint 404s for models published only on the + // global endpoint; retry global once so a stale/ambient region never + // breaks a request that worked before regional routing existed. + const openStream = async (): Promise> => { + if (!plan.fallbackUrl) return openStreamAt(plan.url); + try { + return await openStreamAt(plan.url); + } catch (error) { + if (error instanceof AIError.GoogleApiError && error.status === 404) { + return openStreamAt(plan.fallbackUrl); + } + throw error; + } + }; let body = await openStream(); stream.push({ type: "start", partial: output }); diff --git a/packages/ai/src/providers/google-types.ts b/packages/ai/src/providers/google-types.ts index a5bb299ec..0c5c2fe9f 100644 --- a/packages/ai/src/providers/google-types.ts +++ b/packages/ai/src/providers/google-types.ts @@ -9,9 +9,6 @@ * - `POST {generativelanguage,aiplatform}.googleapis.com/.../models/{model}:streamGenerateContent?alt=sse` * - The Cloud Code Assist endpoint used by `google-gemini-cli.ts` */ - -import type { ServiceTier } from "../types"; - /** Mirror of `@google/genai`'s `FinishReason` string enum. */ export type FinishReason = | "FINISH_REASON_UNSPECIFIED" @@ -137,7 +134,7 @@ export interface GenerateContentConfig { * Gemini/Vertex serving tier. Serialized to the request body root as * `serviceTier` (camelCase) by the transformer in `google-shared.ts`. */ - serviceTier?: ServiceTier; + serviceTier?: "auto" | "default" | "flex" | "scale" | "priority"; abortSignal?: AbortSignal; } diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index 996e763c3..878286a53 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -63,10 +63,19 @@ export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( } if (apiKey) { - const url = `https://aiplatform.googleapis.com/${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + // Explicit `location` is a deliberate residency choice: honor it and let + // a 404 surface. An ambient env-derived region falls back to the global + // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working + // global-only request. + const explicitLocation = options?.location; + const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; + const host = resolveEndpointHost(location); + const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; return { params, - url, + url: `https://${host}/${path}`, + fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, headers: { ...baseHeaders, "x-goog-api-key": apiKey, @@ -110,9 +119,14 @@ function resolveProject(options?: GoogleVertexOptions): string { function resolveEndpointHost(location: string): string { return location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`; } +function resolveAmbientLocation(): string | undefined { + return $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION || undefined; +} +function resolveOptionalLocation(options?: GoogleVertexOptions): string | undefined { + return options?.location || resolveAmbientLocation(); +} function resolveLocation(options?: GoogleVertexOptions): string { - const location = - options?.location || $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION; + const location = resolveOptionalLocation(options); if (!location) { throw new AIError.ConfigurationError( "Vertex AI requires a location. Set GOOGLE_VERTEX_LOCATION/GOOGLE_CLOUD_LOCATION/VERTEX_LOCATION or pass location in options.", From 2b944114fb0c2ae762fb4eff9ab0be957d6fc4a5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 02:48:08 +0000 Subject: [PATCH 29/88] fix(catalog): marked cerebras gemma vision capable Mapped Cerebras gemma-4-31b dynamic discovery to include image input capability and covered the OpenAI Chat Completions image_url serialization path. Fixes #3854 --- ...nai-completions-tool-result-images.test.ts | 43 +++++++++++++++++++ packages/catalog/CHANGELOG.md | 4 ++ .../src/provider-models/openai-compat.ts | 34 ++++++++++++++- .../catalog/test/cerebras-provider.test.ts | 41 ++++++++++++++++++ 4 files changed, 121 insertions(+), 1 deletion(-) create mode 100644 packages/catalog/test/cerebras-provider.test.ts diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts index 6b13398be..bac431780 100644 --- a/packages/ai/test/openai-completions-tool-result-images.test.ts +++ b/packages/ai/test/openai-completions-tool-result-images.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard"; import type { AssistantMessage, Context, Model, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; @@ -75,6 +76,48 @@ function buildToolResult(toolCallId: string, timestamp: number): ToolResultMessa } describe("openai-completions convertMessages", () => { + it("serializes Cerebras gemma image inputs as Chat Completions data URIs", () => { + const model = buildModel({ + id: "gemma-4-31b", + name: "Gemma 4 31B", + api: "openai-completions", + provider: "cerebras", + baseUrl: "https://api.cerebras.ai/v1", + reasoning: false, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 8_192, + }); + const context: Context = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "Identify the shapes and colors. Return JSON only." }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + timestamp: 1, + }, + ], + }; + + const messages = convertMessages(model, context, compat); + + expect(messages).toEqual([ + { + role: "user", + content: [ + { type: "text", text: "Identify the shapes and colors. Return JSON only." }, + { + type: "image_url", + image_url: { url: "data:image/png;base64,ZmFrZQ==" }, + }, + ], + }, + ]); + }); + it("batches tool-result images after consecutive tool results", () => { const baseModel = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">; const model: Model<"openai-completions"> = { diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 09b0580c6..74aa65bd5 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Cerebras `gemma-4-31b` dynamic discovery to mark the model as image-capable so attached images are serialized as OpenAI Chat Completions `image_url` data URIs. ([#3854](https://github.com/can1357/oh-my-pi/issues/3854)) + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 12776ea48..79d32c162 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -803,6 +803,18 @@ export function groqModelManagerOptions(config?: GroqModelManagerConfig): ModelM // 3. Cerebras // --------------------------------------------------------------------------- +const CEREBRAS_IMAGE_INPUT_MODEL_IDS = new Set(["gemma-4-31b"]); + +function applyCerebrasDiscoveryOverrides(model: ModelSpec<"openai-completions">): ModelSpec<"openai-completions"> { + if (!CEREBRAS_IMAGE_INPUT_MODEL_IDS.has(model.id)) { + return model; + } + return { + ...model, + input: ["text", "image"], + }; +} + export interface CerebrasModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -812,7 +824,27 @@ export interface CerebrasModelManagerConfig { export function cerebrasModelManagerOptions( config?: CerebrasModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { - return createSimpleOpenAICompletionsOptions("cerebras", "https://api.cerebras.ai/v1", config); + const apiKey = config?.apiKey; + const baseUrl = config?.baseUrl ?? "https://api.cerebras.ai/v1"; + const references = createBundledReferenceMap<"openai-completions">("cerebras"); + return { + providerId: "cerebras", + ...(apiKey && { + fetchDynamicModels: () => + fetchOpenAICompatibleModels({ + api: "openai-completions", + provider: "cerebras", + baseUrl, + apiKey, + mapModel: (entry, defaults) => { + const reference = references.get(defaults.id); + const model = mapWithBundledReference(entry, defaults, reference); + return applyCerebrasDiscoveryOverrides(model); + }, + fetch: config?.fetch, + }), + }), + }; } // --------------------------------------------------------------------------- diff --git a/packages/catalog/test/cerebras-provider.test.ts b/packages/catalog/test/cerebras-provider.test.ts new file mode 100644 index 000000000..7a829d1bc --- /dev/null +++ b/packages/catalog/test/cerebras-provider.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, test } from "bun:test"; +import { cerebrasModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +describe("Cerebras provider discovery", () => { + test("discovers gemma-4-31b as image-capable", async () => { + const calls: Array<{ url: string; authorization: string | null }> = []; + const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => { + const headers = new Headers(init?.headers); + calls.push({ + url: String(input), + authorization: headers.get("authorization"), + }); + return new Response( + JSON.stringify({ + data: [ + { id: "gemma-4-31b", object: "model" }, + { id: "llama3.1-8b", object: "model" }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }; + + const options = cerebrasModelManagerOptions({ apiKey: "cerebras-test-key", fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + + expect(calls).toEqual([ + { + url: "https://api.cerebras.ai/v1/models", + authorization: "Bearer cerebras-test-key", + }, + ]); + expect(models?.find(model => model.id === "gemma-4-31b")).toMatchObject({ + provider: "cerebras", + api: "openai-completions", + input: ["text", "image"], + }); + expect(models?.find(model => model.id === "llama3.1-8b")?.input).toEqual(["text"]); + }); +}); From be8c2c86a8c796365fa5e70adcfd1a55959db5c0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 02:48:25 +0000 Subject: [PATCH 30/88] style: bun run fix --- .../coding-agent/test/agent-session-handoff.test.ts | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index d8e82aa54..516a4f906 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -452,7 +452,11 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, + { + role: "user", + content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], + timestamp: 1, + }, ], turnPrefixMessages: [], recentMessages: [], @@ -479,7 +483,11 @@ describe("AgentSession handoff", () => { const fixedPreparation: compactionModule.CompactionPreparation = { firstKeptEntryId: lastEntryId, messagesToSummarize: [ - { role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 }, + { + role: "user", + content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], + timestamp: 1, + }, ], turnPrefixMessages: [], recentMessages: [], From 64734021a70df636a597dc858c7366ea0df36ba7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 03:41:24 +0000 Subject: [PATCH 31/88] fix(tui): deliver buffered double-Esc as two events and re-arm loader after task completion MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit StdinBuffer held a bare `\x1b\x1b` chunk (or emitted it as one when followed by a non-CSI byte). `parseKey("\x1b\x1b")` returns undefined, so CustomEditor fell through to the base editor and never fired the configured `onEscape` — the double-escape gesture and the second-press single-Esc handler both went dead whenever the terminal batched the two presses into one stdin read. Split a bare `\x1b\x1b` into two ESC events at the buffer layer, mirroring the existing split for ESC + SGR mouse report. Meta-CSI/SS3 chords (`\x1b\x1b[A`, `\x1b\x1bO…`) still emit as one combined sequence. EventController.tool_execution_update re-armed the working loader when a transient overlay (auto-compaction / auto-retry / handoff) had torn it down mid-tool; tool_execution_end did not. A subagent (`task`) call only fires _end, so a task result landing after such an overlay left the UI looking idle even though the session was still streaming. Mirror the reconciler call in #handleToolExecutionEnd. Fixes #3857 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/controllers/event-controller.ts | 7 ++ .../custom-editor-buffered-double-esc.test.ts | 89 +++++++++++++++++++ .../event-controller-loader-recovery.test.ts | 60 ++++++++++++- packages/tui/CHANGELOG.md | 4 + packages/tui/src/stdin-buffer.ts | 37 ++++++-- packages/tui/test/stdin-buffer.test.ts | 14 +-- 7 files changed, 197 insertions(+), 15 deletions(-) create mode 100644 packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7571e2a01..f6572611c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -21,6 +21,7 @@ - Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error. - Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized. - Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout. +- Fixed the working loader disappearing after a subagent (`task`) tool completed while the focused session was still streaming: `tool_execution_end` did not re-arm the loader the way `tool_execution_update` did, so a tool result landing after a transient overlay (auto-compaction / auto-retry) left the UI looking idle ([#3857](https://github.com/can1357/oh-my-pi/issues/3857)). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 46edcb592..5d3899458 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -890,6 +890,13 @@ export class EventController { } async #handleToolExecutionEnd(event: Extract): Promise { + // A transient overlay (auto-compaction / auto-retry / handoff) that ran + // between this tool's start and end could have detached the working + // loader. `tool_execution_update` already reconciles this so the spinner + // reappears mid-tool; mirror it here so subagent (`task`) completions — + // which only fire `tool_execution_end`, never `_update` — do not leave + // the UI looking idle while the session keeps streaming (#3857). + this.#ensureWorkingLoaderWhileStreaming(); if (event.toolName === "read") { if (this.#inlineReadToolImages(event.toolCallId, event.result)) { const component = this.ctx.pendingTools.get(event.toolCallId); diff --git a/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts b/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts new file mode 100644 index 000000000..1446042fa --- /dev/null +++ b/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts @@ -0,0 +1,89 @@ +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; +import { CustomEditor } from "@oh-my-pi/pi-coding-agent/modes/components/custom-editor"; +import { getEditorTheme, initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { StdinBuffer } from "@oh-my-pi/pi-tui/stdin-buffer"; + +/** + * Regression for #3857. + * + * A fast double-Esc lands as one `"\x1b\x1b"` chunk on stdin. Before the fix, + * `StdinBuffer` either held it as the buffered remainder and timer-flushed it + * as one sequence, or emitted it as one sequence when followed by a non-CSI + * byte. Either way `parseKey("\x1b\x1b")` returns `undefined`, so + * `CustomEditor.handleInput` fell through to the base editor and never fired + * the configured `onEscape` — breaking the double-escape gesture (and any + * single-Esc handler the second press should have hit). + * + * The fix splits a bare `"\x1b\x1b"` into two ESC events at the buffer layer, + * matching the existing split for `"\x1b" + "\x1b[<…"` SGR mouse reports. + */ +describe("buffered double-Esc reaches CustomEditor.onEscape", () => { + beforeAll(async () => { + await initTheme(); + }); + + beforeEach(() => { + vi.useFakeTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + vi.restoreAllMocks(); + }); + + it("fires onEscape twice when a fast double-Esc arrives as one buffered chunk", () => { + const editor = new CustomEditor(getEditorTheme()); + const onEscape = vi.fn(); + editor.onEscape = onEscape; + + const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 }); + buf.on("data", chunk => editor.handleInput(chunk)); + + buf.process("\x1b\x1b"); + // Drain the flush timer chain (main timeout + zero-delay deferral). + vi.runAllTimers(); + + expect(onEscape).toHaveBeenCalledTimes(2); + buf.destroy(); + }); + + it("fires onEscape twice when a double-Esc arrives as one inline chunk followed by a non-CSI byte", () => { + const editor = new CustomEditor(getEditorTheme()); + const onEscape = vi.fn(); + editor.onEscape = onEscape; + + const forwardedToBase: string[] = []; + const baseHandleInput = vi.spyOn(Object.getPrototypeOf(Object.getPrototypeOf(editor)), "handleInput"); + baseHandleInput.mockImplementation(function (this: unknown, data: unknown) { + forwardedToBase.push(data as string); + }); + + const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 }); + buf.on("data", chunk => editor.handleInput(chunk)); + + buf.process("\x1b\x1bX"); + vi.runAllTimers(); + + expect(onEscape).toHaveBeenCalledTimes(2); + // The trailing printable still reaches the base editor in order, after + // both ESC keypresses have fired their handler. + expect(forwardedToBase).toEqual(["X"]); + buf.destroy(); + }); + + it("does not split a meta-CSI arrow into two ESC events", () => { + const editor = new CustomEditor(getEditorTheme()); + const onEscape = vi.fn(); + editor.onEscape = onEscape; + + const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 }); + buf.on("data", chunk => editor.handleInput(chunk)); + + buf.process("\x1b\x1b[A"); + vi.runAllTimers(); + + // alt+up is its own keypress and must never look like two ESC keys. + expect(onEscape).not.toHaveBeenCalled(); + buf.destroy(); + }); +}); diff --git a/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts index 84c01845c..2ab4b0d77 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts @@ -70,7 +70,13 @@ function createContext(options: { terminalProgress?: boolean } = {}) { editor: { getText: () => "" }, sessionManager: { getSessionName: () => "test-session" }, ui: { requestRender: vi.fn(), requestComponentRender: vi.fn(), terminal: { setProgress } }, - viewSession: { isCompacting: false, getLastAssistantMessage: () => undefined }, + viewSession: { + isCompacting: false, + getLastAssistantMessage: () => undefined, + get isStreaming() { + return streamState.isStreaming; + }, + }, session: { get isStreaming() { return streamState.isStreaming; @@ -109,6 +115,15 @@ const RETRY_START = { delayMs: 1000, errorMessage: "overloaded", } as unknown as AgentSessionEvent; +const TASK_TOOL_EXECUTION_END = { + type: "tool_execution_end", + toolCallId: "call-task-1", + toolName: "task", + args: {}, + result: { content: [], details: {} }, + isError: false, +} as unknown as AgentSessionEvent; + describe("EventController loader recovery after overflow maintenance", () => { beforeAll(async () => { @@ -179,6 +194,49 @@ describe("EventController loader recovery after overflow maintenance", () => { expect(statusContainer.children).toContain(ctx.loadingAnimation); }); + it("re-shows the Working… loader after a subagent task completes while the session keeps streaming", async () => { + const { ctx, streamState, statusContainer, workingLoaders } = createContext(); + const controller = new EventController(ctx); + + // Turn begins: the working loader is created and attached. + await controller.handleEvent(AGENT_START); + const firstWorking = workingLoaders[0]; + expect(firstWorking).toBeDefined(); + + // A transient overlay (auto-retry / auto-compaction) tore the loader down + // mid-tool; the session is still streaming when the subagent's task + // completes. Before the fix, `tool_execution_end` (unlike `_update`) did + // not re-arm the loader, so the UI looked idle while the agent kept going. + streamState.isStreaming = true; + ctx.loadingAnimation?.stop(); + ctx.loadingAnimation = undefined; + statusContainer.clear(); + + await controller.handleEvent(TASK_TOOL_EXECUTION_END); + + expect(ctx.loadingAnimation).toBeDefined(); + expect(statusContainer.children).toContain(ctx.loadingAnimation); + expect(workingLoaders).toHaveLength(2); + }); + + it("does not re-arm the Working… loader on tool_execution_end once the session has stopped streaming", async () => { + const { ctx, streamState, statusContainer } = createContext(); + const controller = new EventController(ctx); + + await controller.handleEvent(AGENT_START); + ctx.loadingAnimation?.stop(); + ctx.loadingAnimation = undefined; + statusContainer.clear(); + streamState.isStreaming = false; + + await controller.handleEvent(TASK_TOOL_EXECUTION_END); + + // No streaming → reconciler must stay a no-op; the spinner is not the + // post-turn idle state. + expect(ctx.loadingAnimation).toBeUndefined(); + expect(statusContainer.children).toHaveLength(0); + }); + it("mirrors agent and auto-compaction activity to OSC 9;4 when enabled", async () => { const { ctx, setProgress } = createContext({ terminalProgress: true }); const controller = new EventController(ctx); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index a6873c67d..bee52d7be 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `StdinBuffer` swallowing a fast double-Esc that arrived as one `"\x1b\x1b"` chunk: `parseKey` returns `undefined` for the combined chunk, so the editor's double-escape gesture and any single-Esc handler the second press should have hit never fired. The buffer now splits a bare `"\x1b\x1b"` (whether buffered alone or followed by a non-CSI/SS3 byte) into two ESC events, while meta-CSI/SS3 chords like `"\x1b\x1b[A"` still emit as one sequence ([#3857](https://github.com/can1357/oh-my-pi/issues/3857)). + ## [16.2.3] - 2026-06-28 ### Added diff --git a/packages/tui/src/stdin-buffer.ts b/packages/tui/src/stdin-buffer.ts index e3b155e81..b4f4cb2b8 100644 --- a/packages/tui/src/stdin-buffer.ts +++ b/packages/tui/src/stdin-buffer.ts @@ -241,13 +241,20 @@ function extractCompleteSequences(buffer: string): { sequences: string[]; remain end++; continue; } - // "\x1b\x1b" alone parses as "complete" (legacy alt+esc), but when the - // next byte opens a CSI/SS3 ("[" or "O") this is really ESC prefixing - // another sequence (meta-CSI, or a held Esc keypress joined by a - // follower). Consuming two bytes here would tear the follower and leak - // its tail as typed text (settings search filling with "[B" or - // "[<35;22;17M"). Keep growing; when the buffer ends here, hold the - // partial for the flush window so the disambiguating byte can arrive. + // "\x1b\x1b" is one of three things, and only the first should reach the + // caller as a combined chunk: + // 1. ESC prefixing CSI/SS3 (meta-CSI, held Esc joined by a follower): + // next byte is "[" or "O" — keep growing so the full sequence stays + // together. Consuming two bytes here would tear the follower and + // leak its tail as typed text (settings search filling with "[B" + // or "[<35;22;17M"). + // 2. Two real Esc keypresses bursted by terminal input batching, or + // legacy alt+esc — `parseKey` returns undefined for the combined + // chunk, so a single emission swallows double-escape gestures + // (#3857). Split into two ESC events so the caller observes both. + // When the buffer ends here, hold the partial for the flush window so + // the disambiguating byte (case 1) can still arrive; if it does not, + // the timeout-driven flush splits the held remainder (see `flush`). if (candidate === `${ESC}${ESC}`) { if (end >= length) { return { sequences, remainder: buffer.slice(pos) }; @@ -257,6 +264,10 @@ function extractCompleteSequences(buffer: string): { sequences: string[]; remain end++; continue; } + sequences.push(ESC, ESC); + pos = end; + consumed = true; + break; } // ESC + SGR mouse report is never a meta chord: alt-modified mouse // reports carry the modifier in the button bits, not an ESC prefix. @@ -605,10 +616,18 @@ export class StdinBuffer extends EventEmitter { return []; } - const sequences = [this.#buffer]; + const buffered = this.#buffer; this.#buffer = ""; this.#pendingKittyPrintableCodepoint = undefined; - return sequences; + // Bare double-ESC remainder (no disambiguating "[" / "O" arrived in time): + // two real Esc keypresses bursted by terminal batching, not a meta-CSI/SS3 + // prefix. `parseKey` returns undefined for the combined chunk, so a single + // emission swallows the double-escape gesture (#3857). Mirror the inline + // split in `extractCompleteSequences` and deliver two ESC events. + if (buffered === `${ESC}${ESC}`) { + return [ESC, ESC]; + } + return [buffered]; } clear(): void { diff --git a/packages/tui/test/stdin-buffer.test.ts b/packages/tui/test/stdin-buffer.test.ts index 2a624fb7d..efb39aa9a 100644 --- a/packages/tui/test/stdin-buffer.test.ts +++ b/packages/tui/test/stdin-buffer.test.ts @@ -181,16 +181,20 @@ describe("StdinBuffer", () => { expect(emittedSequences).toEqual(["\x1b", "\x1b[<35;22;17M"]); }); - it("flushes a trailing double-ESC as one sequence after the timeout", async () => { + it("splits a trailing double-ESC into two ESC events after the timeout", async () => { + // A bare `\x1b\x1b` is two real Esc keypresses (or legacy alt+esc). + // `parseKey` returns undefined for the combined chunk, so emitting it + // as one swallows double-escape gestures (#3857). Split on flush so + // downstream handlers fire twice. processInput("\x1b\x1b"); expect(emittedSequences).toEqual([]); - await waitUntil(() => emittedSequences.length > 0); - expect(emittedSequences).toEqual(["\x1b\x1b"]); + await waitUntil(() => emittedSequences.length >= 2); + expect(emittedSequences).toEqual(["\x1b", "\x1b"]); }); - it("keeps double-ESC followed by a non-CSI byte split as before", () => { + it("splits double-ESC followed by a non-CSI byte into two ESC events plus the byte", () => { processInput("\x1b\x1bX"); - expect(emittedSequences).toEqual(["\x1b\x1b", "X"]); + expect(emittedSequences).toEqual(["\x1b", "\x1b", "X"]); }); it("consumes a whole meta-CSI arrow in one chunk", () => { From 7871591b4ac25954ff3af9a2c30c982b19619ada Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 03:40:53 +0000 Subject: [PATCH 32/88] fix(coding-agent): restored working loader after subagent ends inside overlay window MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `tool_execution_end` for a long-running tool (`task` subagent, async bash poll, …) is the next streaming event on the parent session when an inner transient overlay (auto-snapcompact, auto-context-full, auto-retry) nulled the working loader between the tool's start and end. The overlay-end handlers are the only loader restorers keyed off the missing reference; if the subagent's `tool_execution_end` lands while the overlay is still active (or its end handler errored before re-arming), the spinner stays gone for the rest of the parent turn even though the agent keeps streaming. `#handleToolExecutionEnd` now calls `#ensureWorkingLoaderWhileStreaming()` at the top, mirroring `tool_execution_update` so the working loader survives a subagent completing inside the overlay window. Refs #3858 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/controllers/event-controller.ts | 11 +++++ .../event-controller-error-banner.test.ts | 41 +++++++++++++++++++ .../event-controller-todo-reminder.test.ts | 5 +++ 4 files changed, 58 insertions(+) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7571e2a01..3d9100e83 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -21,6 +21,7 @@ - Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error. - Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized. - Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout. +- Fixed the `Working…` loader staying gone for the rest of a parent turn when a long-running tool (e.g. a `task` subagent) finished inside a transient overlay window (auto-snapcompact, auto-context-full, auto-retry). Those overlays null the working loader on start and the overlay-end handler is the only restorer keyed off the missing loader; if the subagent's `tool_execution_end` lands while the overlay is still active (or its end handler errored before re-arming), the spinner stayed gone until the next turn even though the parent kept streaming. `tool_execution_end` now mirrors the `tool_execution_update` self-heal so the working loader survives a subagent completing inside the overlay window ([#3858](https://github.com/can1357/oh-my-pi/issues/3858)). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 46edcb592..b9673169d 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -890,6 +890,17 @@ export class EventController { } async #handleToolExecutionEnd(event: Extract): Promise { + // A long-running tool (`task` subagent, async bash poll, …) is the next + // streaming event on the parent session when an inner transient overlay + // (auto-compaction / auto-retry) clears the status container between this + // tool's start and end. `auto_*_end` is the overlay-side restorer; for any + // path that nulls the loader without a follow-up overlay-end (e.g. an + // errored overlay teardown, or the subagent finishing inside the overlay + // window), the next tool event is the only restore point before the turn + // settles. `tool_execution_update` already self-heals — keep + // `tool_execution_end` symmetric so the spinner survives a subagent + // completing inside the overlay window (#3858). + this.#ensureWorkingLoaderWhileStreaming(); if (event.toolName === "read") { if (this.#inlineReadToolImages(event.toolCallId, event.result)) { const component = this.ctx.pendingTools.get(event.toolCallId); diff --git a/packages/coding-agent/test/event-controller-error-banner.test.ts b/packages/coding-agent/test/event-controller-error-banner.test.ts index 97284b7df..eb982eaa4 100644 --- a/packages/coding-agent/test/event-controller-error-banner.test.ts +++ b/packages/coding-agent/test/event-controller-error-banner.test.ts @@ -299,6 +299,47 @@ describe("EventController working loader reconciliation", () => { expect(ctx.ensureLoadingAnimation).toHaveBeenCalledTimes(1); }); + it("self-heals missing working loader when a task subagent finishes mid-turn (#3858)", async () => { + // `task` subagents run inside the parent's streaming turn. While the task is + // running a transient overlay (auto-compaction / auto-retry) can drop the + // working loader by clearing the status container, and the overlay's end + // handler is the only restorer keyed off the missing loader. If the task + // finishes between the overlay's start and end (or any other branch where + // the loader was nulled without a follow-up overlay-end), `tool_execution_end` + // is the next streaming event that lands and must heal the loader, mirroring + // the `tool_execution_update` reconciler. Without this the spinner stays + // gone for the remainder of the parent turn even though the agent keeps + // streaming (the user-visible regression in #3858). + const { controller, ctx } = createFixture(); + (ctx.viewSession as unknown as { isStreaming: boolean }).isStreaming = true; + + await controller.handleEvent({ + type: "tool_execution_end", + toolCallId: "task-1", + toolName: "task", + isError: false, + result: { content: [{ type: "text", text: "ok" }], details: {} }, + } as Extract); + + expect(ctx.ensureLoadingAnimation).toHaveBeenCalledTimes(1); + }); + + it("does not restore the working loader while an overlay loader (auto-retry) owns the status container at tool_execution_end", async () => { + const { controller, ctx } = createFixture(); + ctx.retryLoader = { stop: vi.fn() } as unknown as InteractiveModeContext["retryLoader"]; + (ctx.viewSession as unknown as { isStreaming: boolean }).isStreaming = true; + + await controller.handleEvent({ + type: "tool_execution_end", + toolCallId: "task-2", + toolName: "task", + isError: false, + result: { content: [{ type: "text", text: "ok" }], details: {} }, + } as Extract); + + expect(ctx.ensureLoadingAnimation).not.toHaveBeenCalled(); + }); + it("keeps transient retry status exclusive while a retry loader is visible", async () => { const { controller, ctx } = createFixture(); ctx.retryLoader = { stop: vi.fn() } as unknown as InteractiveModeContext["retryLoader"]; diff --git a/packages/coding-agent/test/event-controller-todo-reminder.test.ts b/packages/coding-agent/test/event-controller-todo-reminder.test.ts index 15b1b5cbc..666b74fe4 100644 --- a/packages/coding-agent/test/event-controller-todo-reminder.test.ts +++ b/packages/coding-agent/test/event-controller-todo-reminder.test.ts @@ -21,6 +21,11 @@ function createContext() { updateEditorTopBorder: vi.fn(), clearPinnedError: vi.fn(), ensureLoadingAnimation: vi.fn(), + // `viewSession.isStreaming` is read by `#ensureWorkingLoaderWhileStreaming`, + // which runs at the top of `tool_execution_end` (and other streaming-event + // handlers). Leaving it false matches the implicit assumption in this + // fixture: the todo HUD lifecycle is independent of the working loader. + viewSession: { isStreaming: false }, todoReminderContainer, setTodos: vi.fn(), present, From c2d6c8cf1850aa57a41e6d104f5a704cf9b5149f Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 03:41:37 +0000 Subject: [PATCH 33/88] fix(tui): deliver buffered double-Esc as two events and re-arm loader after task completion MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit StdinBuffer held a bare `\x1b\x1b` chunk and timer-flushed it as one sequence. `parseKey("\x1b\x1b")` returns undefined, so CustomEditor fell through to the base editor and never fired the configured `onEscape` — the double-escape gesture and the second-press single-Esc handler both went dead whenever the terminal batched the two presses into one stdin read. Split an exact bare `\x1b\x1b` into two ESC events only after the flush window proves no follower arrived. If a follower does arrive, emit the first ESC and restart parsing at the second ESC so legacy Alt chords (`\x1bd`, `\x1b\x7f`) remain one downstream keypress. Meta-CSI/SS3 chords (`\x1b\x1b[A`, `\x1b\x1bO…`) still emit as one combined sequence. EventController.tool_execution_update re-armed the working loader when a transient overlay (auto-compaction / auto-retry / handoff) had torn it down mid-tool; tool_execution_end did not. A subagent (`task`) call only fires _end, so a task result landing after such an overlay left the UI looking idle even though the session was still streaming. Mirror the reconciler call in #handleToolExecutionEnd. Fixes #3857 --- .../custom-editor-buffered-double-esc.test.ts | 30 +++++++------------ .../event-controller-loader-recovery.test.ts | 1 - packages/tui/CHANGELOG.md | 2 +- packages/tui/src/stdin-buffer.ts | 21 +++++++------ packages/tui/test/stdin-buffer.test.ts | 12 ++++++-- 5 files changed, 32 insertions(+), 34 deletions(-) diff --git a/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts b/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts index 1446042fa..8e518ef53 100644 --- a/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts +++ b/packages/coding-agent/test/custom-editor-buffered-double-esc.test.ts @@ -7,15 +7,14 @@ import { StdinBuffer } from "@oh-my-pi/pi-tui/stdin-buffer"; * Regression for #3857. * * A fast double-Esc lands as one `"\x1b\x1b"` chunk on stdin. Before the fix, - * `StdinBuffer` either held it as the buffered remainder and timer-flushed it - * as one sequence, or emitted it as one sequence when followed by a non-CSI - * byte. Either way `parseKey("\x1b\x1b")` returns `undefined`, so + * `StdinBuffer` held it as the buffered remainder, then timer-flushed it as + * one sequence. `parseKey("\x1b\x1b")` returns `undefined`, so * `CustomEditor.handleInput` fell through to the base editor and never fired - * the configured `onEscape` — breaking the double-escape gesture (and any - * single-Esc handler the second press should have hit). + * the configured `onEscape` — breaking the double-escape gesture. * - * The fix splits a bare `"\x1b\x1b"` into two ESC events at the buffer layer, - * matching the existing split for `"\x1b" + "\x1b[<…"` SGR mouse reports. + * The fix splits a bare `"\x1b\x1b"` into two ESC events only when no follower + * arrives in the disambiguation window. If a follower arrives, the second ESC + * remains attached to that follower so legacy Alt chords survive. */ describe("buffered double-Esc reaches CustomEditor.onEscape", () => { beforeAll(async () => { @@ -47,27 +46,20 @@ describe("buffered double-Esc reaches CustomEditor.onEscape", () => { buf.destroy(); }); - it("fires onEscape twice when a double-Esc arrives as one inline chunk followed by a non-CSI byte", () => { + it("preserves a legacy Alt chord batched after a bare ESC", () => { const editor = new CustomEditor(getEditorTheme()); const onEscape = vi.fn(); editor.onEscape = onEscape; - - const forwardedToBase: string[] = []; - const baseHandleInput = vi.spyOn(Object.getPrototypeOf(Object.getPrototypeOf(editor)), "handleInput"); - baseHandleInput.mockImplementation(function (this: unknown, data: unknown) { - forwardedToBase.push(data as string); - }); + editor.setText("foo bar"); const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 }); buf.on("data", chunk => editor.handleInput(chunk)); - buf.process("\x1b\x1bX"); + buf.process("\x1b\x1b\x7f"); vi.runAllTimers(); - expect(onEscape).toHaveBeenCalledTimes(2); - // The trailing printable still reaches the base editor in order, after - // both ESC keypresses have fired their handler. - expect(forwardedToBase).toEqual(["X"]); + expect(onEscape).toHaveBeenCalledTimes(1); + expect(editor.getText()).toBe("foo "); buf.destroy(); }); diff --git a/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts index 2ab4b0d77..b5e1b1109 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-loader-recovery.test.ts @@ -124,7 +124,6 @@ const TASK_TOOL_EXECUTION_END = { isError: false, } as unknown as AgentSessionEvent; - describe("EventController loader recovery after overflow maintenance", () => { beforeAll(async () => { await initTheme(false); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index bee52d7be..cf0d338cb 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `StdinBuffer` swallowing a fast double-Esc that arrived as one `"\x1b\x1b"` chunk: `parseKey` returns `undefined` for the combined chunk, so the editor's double-escape gesture and any single-Esc handler the second press should have hit never fired. The buffer now splits a bare `"\x1b\x1b"` (whether buffered alone or followed by a non-CSI/SS3 byte) into two ESC events, while meta-CSI/SS3 chords like `"\x1b\x1b[A"` still emit as one sequence ([#3857](https://github.com/can1357/oh-my-pi/issues/3857)). +- Fixed `StdinBuffer` swallowing a fast double-Esc that arrived as one `"\x1b\x1b"` chunk: `parseKey` returns `undefined` for the combined chunk, so the editor's double-escape gesture and any single-Esc handler the second press should have hit never fired. The buffer now splits a bare `"\x1b\x1b"` into two ESC events only when no follower arrives in the disambiguation window; when a follower arrives, the second ESC stays attached so legacy Alt chords like `"\x1bd"` and meta-CSI/SS3 chords like `"\x1b\x1b[A"` still emit as parseable sequences ([#3857](https://github.com/can1357/oh-my-pi/issues/3857)). ## [16.2.3] - 2026-06-28 diff --git a/packages/tui/src/stdin-buffer.ts b/packages/tui/src/stdin-buffer.ts index b4f4cb2b8..e54b57647 100644 --- a/packages/tui/src/stdin-buffer.ts +++ b/packages/tui/src/stdin-buffer.ts @@ -241,20 +241,19 @@ function extractCompleteSequences(buffer: string): { sequences: string[]; remain end++; continue; } - // "\x1b\x1b" is one of three things, and only the first should reach the - // caller as a combined chunk: + // "\x1b\x1b" is one of three things: // 1. ESC prefixing CSI/SS3 (meta-CSI, held Esc joined by a follower): // next byte is "[" or "O" — keep growing so the full sequence stays // together. Consuming two bytes here would tear the follower and // leak its tail as typed text (settings search filling with "[B" // or "[<35;22;17M"). - // 2. Two real Esc keypresses bursted by terminal input batching, or - // legacy alt+esc — `parseKey` returns undefined for the combined - // chunk, so a single emission swallows double-escape gestures - // (#3857). Split into two ESC events so the caller observes both. - // When the buffer ends here, hold the partial for the flush window so - // the disambiguating byte (case 1) can still arrive; if it does not, - // the timeout-driven flush splits the held remainder (see `flush`). + // 2. ESC followed by a legacy Alt chord (`\x1bd`, `\x1b\x7f`, …): + // emit the first ESC, then restart at the second ESC so downstream + // parsing still sees the Alt chord as one keypress (#3860 review). + // 3. Two real Esc keypresses bursted by terminal input batching: + // when the buffer ends here, hold the partial for the flush window + // so case 1/2 can still arrive; if no follower arrives, `flush()` + // splits the held remainder into two ESC events (#3857). if (candidate === `${ESC}${ESC}`) { if (end >= length) { return { sequences, remainder: buffer.slice(pos) }; @@ -264,8 +263,8 @@ function extractCompleteSequences(buffer: string): { sequences: string[]; remain end++; continue; } - sequences.push(ESC, ESC); - pos = end; + sequences.push(ESC); + pos += 1; consumed = true; break; } diff --git a/packages/tui/test/stdin-buffer.test.ts b/packages/tui/test/stdin-buffer.test.ts index efb39aa9a..c7fa0bda4 100644 --- a/packages/tui/test/stdin-buffer.test.ts +++ b/packages/tui/test/stdin-buffer.test.ts @@ -192,9 +192,17 @@ describe("StdinBuffer", () => { expect(emittedSequences).toEqual(["\x1b", "\x1b"]); }); - it("splits double-ESC followed by a non-CSI byte into two ESC events plus the byte", () => { + it("preserves legacy Alt chords batched after a bare ESC", () => { processInput("\x1b\x1bX"); - expect(emittedSequences).toEqual(["\x1b", "\x1b", "X"]); + expect(emittedSequences).toEqual(["\x1b", "\x1bX"]); + + emittedSequences = []; + processInput("\x1b\x1bd"); + expect(emittedSequences).toEqual(["\x1b", "\x1bd"]); + + emittedSequences = []; + processInput("\x1b\x1b\x7f"); + expect(emittedSequences).toEqual(["\x1b", "\x1b\x7f"]); }); it("consumes a whole meta-CSI arrow in one chunk", () => { From b87110c66b920ed334f307691b8b71fb07fb91e5 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 06:34:07 +0200 Subject: [PATCH 34/88] fix(ai): preferred explicit env api key over stored static key MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Reordered `AuthStorage.getApiKey` and its async session variant to resolve stored OAuth, then an env var, then a stored static api_key — so an explicit `GEMINI_API_KEY`-style env var now wins over a stale broker-migrated key. - Reworked `getCredentialOrigin` and `getApiKeySource` to mirror that precedence via a `describeStored` helper, reporting `oauth`/`env`/`api_key` in the new order. - Updated `auth-storage-api-key-login` to neutralize ambient env keys with a `getEnvApiKey` spy, and `auth-storage-credential-origin` to assert OAuth outranks api_key and env outranks a stored api_key. --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/auth-storage.ts | 108 +++++++++--------- .../test/auth-storage-api-key-login.test.ts | 5 + .../auth-storage-credential-origin.test.ts | 16 ++- 4 files changed, 75 insertions(+), 55 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 5e64524fc..7d59269e4 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -14,6 +14,7 @@ - Updated internal `coerceServiceTierByFamily` helper to facilitate migration from legacy settings ### Fixed +- Changed API-key resolution precedence so an explicit environment variable (e.g. `GEMINI_API_KEY`) overrides a stored/broker-migrated static API key; a deliberate OAuth login still takes precedence over the env var. - Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 6060fcb76..d6d91c9f0 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -1736,8 +1736,8 @@ export class AuthStorage { /** * Classify where a provider's auth comes from, following the same precedence * as {@link AuthStorage.getApiKey}: runtime override → config override → - * stored credential (api_key before oauth, matching getApiKey) → env var → - * fallback resolver. Returns undefined when no auth is configured. + * stored OAuth → env var → stored api_key → fallback resolver. Returns + * undefined when no auth is configured. * * Compact, structured counterpart to {@link describeCredentialSource}. */ @@ -1745,10 +1745,9 @@ export class AuthStorage { if (this.#runtimeOverrides.has(provider)) return { kind: "runtime" }; if (this.#configOverrides.has(provider)) return { kind: "config" }; const stored = this.#getCredentialsForProvider(provider); - if (stored.length > 0) { - return { kind: stored.some(credential => credential.type === "api_key") ? "api_key" : "oauth" }; - } + if (stored.some(credential => credential.type === "oauth")) return { kind: "oauth" }; if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) }; + if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" }; if (this.#fallbackResolver?.(provider)) return { kind: "fallback" }; return undefined; } @@ -3760,12 +3759,8 @@ export class AuthStorage { return configKey; } - const apiKeySelection = this.#selectCredentialByType(provider, "api_key"); - if (apiKeySelection) { - return this.#configValueResolver(apiKeySelection.credential.key); - } - - // Return current OAuth access token only if it is not already expired. + // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored + // static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthSelection = this.#selectCredentialByType(provider, "oauth"); if (oauthSelection) { const expiresAt = oauthSelection.credential.expires; @@ -3784,17 +3779,22 @@ export class AuthStorage { const envKey = getEnvApiKey(provider); if (envKey) return envKey; + const apiKeySelection = this.#selectCredentialByType(provider, "api_key"); + if (apiKeySelection) { + return this.#configValueResolver(apiKeySelection.credential.key); + } + return this.#fallbackResolver?.(provider) ?? undefined; } /** * Get API key for a provider. - * Priority: + * Priority (first match wins): * 1. Runtime override (CLI --api-key) * 2. Config override (models.yml `providers..apiKey`) - * 3. API key from storage - * 4. OAuth token from storage (auto-refreshed) - * 5. Environment variable + * 3. OAuth token from storage (auto-refreshed) + * 4. Environment variable + * 5. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins * 6. Fallback resolver (models.yml custom providers, last-resort) */ async getApiKey(provider: string, sessionId?: string, options?: AuthApiKeyOptions): Promise { @@ -3814,25 +3814,27 @@ export class AuthStorage { return configKey; } - const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId); - if (apiKeySelection) { - this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index); - return this.#configValueResolver(apiKeySelection.credential.key); - } - + // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored + // static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthResolved = await this.#resolveOAuthSelection(provider, sessionId, options); if (oauthResolved) { return oauthResolved.apiKey; } - // Fall back to environment variable or custom resolver. If we reach here after - // an OAuth miss, the session sticky (if any) is stale — the request will - // authenticate via env/fallback, not OAuth, so clear the sticky now so that - // getOAuthAccountId() correctly suppresses account_uuid for this session. + // Past OAuth: the session sticky (if any) is stale — the request authenticates via + // env/api_key/fallback, not OAuth, so clear it now so getOAuthAccountId() correctly + // suppresses account_uuid for this session. if (sessionId) this.#sessionLastCredential.get(provider)?.delete(sessionId); + const envKey = getEnvApiKey(provider); if (envKey) return envKey; + const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId); + if (apiKeySelection) { + this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index); + return this.#configValueResolver(apiKeySelection.credential.key); + } + // Fall back to custom resolver (e.g., models.json custom providers) return this.#fallbackResolver?.(provider) ?? undefined; } @@ -4486,12 +4488,13 @@ export class AuthStorage { /** * Describe where the active credential for a provider came from. * - * Surfaces four layers, highest precedence first: + * Mirrors {@link AuthStorage.getApiKey} precedence, highest first: * 1. Runtime override (`--api-key`). * 2. Config override (`models.yml` `providers..apiKey`). - * 3. Stored credential (the one this session is currently sticky to, or the - * one round-robin would pick next when no session id is supplied). - * 4. Env var / fallback resolver — when no stored credential exists. + * 3. Stored OAuth credential. + * 4. Env var — overrides a stored static api_key (e.g. a stale broker copy). + * 5. Stored api_key credential. + * 6. Fallback resolver. * * The string is purely informational; consumers must not parse it. */ @@ -4505,30 +4508,31 @@ export class AuthStorage { const baseLabel = this.#sourceLabel ?? "local store"; const stored = this.#getStoredCredentials(provider); - if (stored.length === 0) { - if (getEnvApiKey(provider)) return `env ${baseLabel ? `(fallback over ${baseLabel})` : ""}`.trim(); - if (this.#fallbackResolver?.(provider) !== undefined) return `fallback resolver`; - return undefined; - } - const session = sessionId ? this.#sessionLastCredential.get(provider)?.get(sessionId) : undefined; - // Same selection logic as #selectCredentialByType for "no session" lookups: prefer - // the type with stored credentials, lean OAuth before api_key. We don't run the - // full round-robin here because describing the source shouldn't advance the index. - const preferredType: AuthCredential["type"] = - session?.type ?? (stored.some(entry => entry.credential.type === "oauth") ? "oauth" : "api_key"); - const typed = stored - .map((entry, index) => ({ entry, index })) - .filter(({ entry }) => entry.credential.type === preferredType); - if (typed.length === 0) return baseLabel; - const index = session?.index ?? typed[0].index; - const chosen = stored[index] ?? typed[0].entry; - const credential = chosen.credential; - const identity = - credential.type === "oauth" - ? (credential.email ?? credential.accountId ?? credential.projectId ?? `cred ${chosen.id}`) - : `cred ${chosen.id}`; - return `${baseLabel} · ${preferredType} #${chosen.id} (${identity})`; + // Describe the stored credential of a given type, honoring the session sticky index. + const describeStored = (type: AuthCredential["type"]): string | undefined => { + const typed = stored + .map((entry, index) => ({ entry, index })) + .filter(({ entry }) => entry.credential.type === type); + if (typed.length === 0) return undefined; + const index = session?.type === type ? session.index : typed[0].index; + const chosen = stored[index] ?? typed[0].entry; + const credential = chosen.credential; + const identity = + credential.type === "oauth" + ? (credential.email ?? credential.accountId ?? credential.projectId ?? `cred ${chosen.id}`) + : `cred ${chosen.id}`; + return `${baseLabel} · ${type} #${chosen.id} (${identity})`; + }; + + // A deliberate OAuth login wins; then an explicit env var; then a stored static api_key. + const oauthSource = describeStored("oauth"); + if (oauthSource) return oauthSource; + if (getEnvApiKey(provider)) return `env (over ${baseLabel})`; + const apiKeySource = describeStored("api_key"); + if (apiKeySource) return apiKeySource; + if (this.#fallbackResolver?.(provider) !== undefined) return "fallback resolver"; + return undefined; } } diff --git a/packages/ai/test/auth-storage-api-key-login.test.ts b/packages/ai/test/auth-storage-api-key-login.test.ts index e50e0fd4f..879a8a9ec 100644 --- a/packages/ai/test/auth-storage-api-key-login.test.ts +++ b/packages/ai/test/auth-storage-api-key-login.test.ts @@ -8,6 +8,7 @@ import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai/auth-sto import * as deepseekModule from "@oh-my-pi/pi-ai/registry/deepseek"; import * as kagiModule from "@oh-my-pi/pi-ai/registry/kagi"; import * as ollamaCloudModule from "@oh-my-pi/pi-ai/registry/ollama-cloud"; +import * as aiStream from "@oh-my-pi/pi-ai/stream"; import { removeWithRetries } from "../../utils/src/temp"; function countCredentialRows(dbPath: string, provider: string): number { @@ -38,6 +39,9 @@ function countCredentialRowsByDisabledState(dbPath: string, provider: string, di } describe("AuthStorage api-key login upsert", () => { + // A live env var now (correctly) overrides a stored static api_key. These tests verify that a + // freshly stored api_key resolves through AuthStorage.getApiKey, so neutralize the env leg + // entirely — this ignores every provider's ambient env key, not just the few set locally. let tempDir = ""; let dbPath = ""; let store: SqliteAuthCredentialStore | null = null; @@ -47,6 +51,7 @@ describe("AuthStorage api-key login upsert", () => { let loginOllamaCloudSpy: Mock; beforeEach(async () => { + vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-api-key-login-")); dbPath = path.join(tempDir, "agent.db"); store = await SqliteAuthCredentialStore.open(dbPath); diff --git a/packages/ai/test/auth-storage-credential-origin.test.ts b/packages/ai/test/auth-storage-credential-origin.test.ts index 822ffeff4..e3d7b2d10 100644 --- a/packages/ai/test/auth-storage-credential-origin.test.ts +++ b/packages/ai/test/auth-storage-credential-origin.test.ts @@ -68,14 +68,24 @@ describe("AuthStorage.getCredentialOrigin", () => { }); }); - test("a stored api key reports api_key and outranks a co-stored OAuth credential", async () => { + test("a stored OAuth credential outranks a co-stored api key", async () => { await withEnv(SUPPRESS_ENV, async () => { - // getApiKey() prefers api_key before oauth, so the origin must match. + // getApiKey() resolves stored OAuth before a stored api_key, so the origin must match. await auth?.set("openai", [ { type: "oauth", access: "a", refresh: "r", expires: Date.now() + 60_000 }, { type: "api_key", key: "sk-stored" }, ]); - expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "api_key" }); + expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "oauth" }); + }); + }); + + test("an explicit env var outranks a stored api key", async () => { + // Regression: a live env var is the user's current choice and must win over a stored + // static api_key (e.g. a stale broker-migrated copy) so `GEMINI_API_KEY` etc. take effect. + await withEnv({ ...SUPPRESS_ENV, OPENAI_API_KEY: "sk-env" }, async () => { + await auth?.set("openai", [{ type: "api_key", key: "sk-stored" }]); + expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "env", envVar: "OPENAI_API_KEY" }); + expect(await auth?.getApiKey("openai")).toBe("sk-env"); }); }); From d01bf079ef24f736514089c4a43521f3ab15a138 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 06:34:31 +0200 Subject: [PATCH 35/88] feat(ai): added google interactions api transport - Added `google-interactions.ts` streaming the Gemini Interactions API (model-mode steps, thought/tool/text deltas, usage, `previous_interaction_id` lineage) shared by the direct Google and Vertex providers. - Routed `streamGoogle` and `streamGoogleVertex` through `resolveInteractionDispatch`, defaulting Gemini 3+ onto Interactions (official endpoint for direct Google, bearer/ADC for Vertex) with transparent `:streamGenerateContent` fallback and a `useInteractionsApi: false` opt-out. - Added explicit Vertex bearer support in `google-auth.ts` via `GOOGLE_CLOUD_ACCESS_TOKEN`/`CLOUDSDK_AUTH_ACCESS_TOKEN` plus a `hasVertexBearerCredentialsHint` probe gating the auto-default. - Added `useInteractionsApi`/`storeInteraction`/`previousInteractionId` to `StreamOptions` and `GoogleSharedStreamOptions`, threading them through `mapOptionsForApi`. - Added `google-interactions` coverage and pinned the existing generateContent assertions with `useInteractionsApi: false`. --- packages/ai/CHANGELOG.md | 2 + packages/ai/src/providers/google-auth.ts | 25 + .../ai/src/providers/google-interactions.ts | 753 ++++++++++++++++++ packages/ai/src/providers/google-shared.ts | 13 + packages/ai/src/providers/google-vertex.ts | 173 ++-- packages/ai/src/providers/google.ts | 80 +- packages/ai/src/stream.ts | 3 + packages/ai/src/types.ts | 16 + .../test/google-empty-response-retry.test.ts | 7 +- packages/ai/test/google-interactions.test.ts | 511 ++++++++++++ packages/ai/test/google-service-tier.test.ts | 6 +- packages/ai/test/google-system-prompt.test.ts | 2 + packages/ai/test/issue-1270-repro.test.ts | 2 + 13 files changed, 1507 insertions(+), 86 deletions(-) create mode 100644 packages/ai/src/providers/google-interactions.ts create mode 100644 packages/ai/test/google-interactions.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7d59269e4..ae67a9d21 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -6,6 +6,8 @@ - Added service tier support for Google Gemini and Vertex AI - Introduced `ServiceTierByFamily` to allow model-specific service tier configurations +- Added Google Vertex AI Interactions API support, sharing one Interactions transport across the direct Google and Vertex providers. Gemini 3+ models use the Interactions API by default — on the official `generativelanguage` endpoint for direct Google, and under ADC/bearer auth for Vertex — with automatic fallback to `:streamGenerateContent` when the model/endpoint can't serve Interactions. Pass `useInteractionsApi: false` to force generateContent. +- Added support for an explicit Vertex bearer access token via `GOOGLE_CLOUD_ACCESS_TOKEN` / `CLOUDSDK_AUTH_ACCESS_TOKEN`, so `gcloud auth print-access-token` can drive Vertex without a full `application-default login`. ### Changed diff --git a/packages/ai/src/providers/google-auth.ts b/packages/ai/src/providers/google-auth.ts index 11a2cd6c4..7aabaf5cc 100644 --- a/packages/ai/src/providers/google-auth.ts +++ b/packages/ai/src/providers/google-auth.ts @@ -13,6 +13,7 @@ */ import { Buffer } from "node:buffer"; +import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { $envpos, isEnoent, logger } from "@oh-my-pi/pi-utils"; @@ -281,6 +282,11 @@ const SHARED_TOKEN_RESOLVE_TIMEOUT_MS = 30_000; * The token is cached in module scope and refreshed `GOOGLE_VERTEX_REFRESH_SKEW_MS` ms before it expires. */ export async function getVertexAccessToken(options?: { signal?: AbortSignal; fetch?: FetchImpl }): Promise { + // An explicit access token (e.g. `gcloud auth print-access-token`) bypasses the cache so a + // refreshed env token takes effect immediately. `CLOUDSDK_AUTH_ACCESS_TOKEN` is gcloud's own + // override var; `GOOGLE_CLOUD_ACCESS_TOKEN` is the omp-facing alias. + const explicitToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN; + if (explicitToken) return explicitToken; const fetchImpl = options?.fetch ?? globalThis.fetch.bind(globalThis); const skew = getRefreshSkewMs(); const now = Date.now(); @@ -323,3 +329,22 @@ export function __resetVertexTokenCache(): void { tokenCache.clear(); inflight.clear(); } + +/** + * Sync best-effort probe for a usable Vertex bearer credential source — an explicit access-token + * env var, `GOOGLE_APPLICATION_CREDENTIALS`, a user ADC file, or a GCP runtime whose metadata + * server can mint ADC (GCE/Cloud Run/App Engine/Functions). Lets callers prefer the bearer + * Interactions transport only when ADC is actually reachable, without paying the async + * metadata-probe cost for API-key-only setups. + */ +export function hasVertexBearerCredentialsHint(): boolean { + if (Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN) return true; + if (Bun.env.GOOGLE_APPLICATION_CREDENTIALS) return true; + // GCP-hosted runtimes expose ADC via the metadata server; these env vars mark those runtimes. + if (Bun.env.K_SERVICE || Bun.env.FUNCTION_TARGET || Bun.env.GAE_ENV || Bun.env.GCE_METADATA_HOST) return true; + try { + return fs.existsSync(userAdcPath()); + } catch { + return false; + } +} diff --git a/packages/ai/src/providers/google-interactions.ts b/packages/ai/src/providers/google-interactions.ts new file mode 100644 index 000000000..3f5612809 --- /dev/null +++ b/packages/ai/src/providers/google-interactions.ts @@ -0,0 +1,753 @@ +import { parseGeminiModel } from "@oh-my-pi/pi-catalog/identity"; +import { calculateCost } from "@oh-my-pi/pi-catalog/models"; +import { fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils"; +import * as AIError from "../error"; +import type { + AssistantMessage, + Context, + FetchImpl, + ImageContent, + Message, + Model, + ProviderSessionState, + TextContent, + ToolCall, + Usage, +} from "../types"; +import { shouldSendServiceTier } from "../types"; +import { normalizeSystemPrompts } from "../utils"; +import { AssistantMessageEventStream } from "../utils/event-stream"; +import { convertTools, type GoogleSharedStreamOptions, type GoogleThinkingLevel } from "./google-shared"; + +type GoogleInteractionsApi = "google-generative-ai" | "google-vertex"; +type GoogleInteractionsModel = Model; +type GoogleOptions = GoogleSharedStreamOptions; + +const GOOGLE_INTERACTIONS_STATE_KEY = "google-interactions-state"; + +/** Provider session state storing the last Gemini Interactions response id. */ +export interface GoogleInteractionsProviderSessionState extends ProviderSessionState { + lastInteractionId?: string; +} + +/** Conversation anchor for continuing an Interactions turn from a prior assistant response. */ +export interface InteractionAnchor { + id?: string; + messageIndex?: number; +} + +type InteractionContent = { type: "text"; text: string } | { type: "image"; data: string; mime_type: string }; + +interface InteractionUserInputStep { + type: "user_input"; + content: InteractionContent[]; +} + +interface InteractionModelOutputStep { + type: "model_output"; + content: InteractionContent[]; +} + +interface InteractionFunctionCallStep { + type: "function_call"; + id: string; + name: string; + arguments: Record; +} + +interface InteractionFunctionResultStep { + type: "function_result"; + name: string; + call_id: string; + result: InteractionContent[]; + is_error?: boolean; +} + +interface InteractionThoughtStep { + type: "thought"; + summary?: InteractionContent[]; + signature?: string; +} + +interface InteractionThoughtSummaryDelta { + type: "thought_summary"; + content?: InteractionContent; +} + +interface InteractionThoughtSignatureDelta { + type: "thought_signature"; + signature?: string; +} + +interface InteractionArgumentsDelta { + type: "arguments_delta"; + arguments?: string; +} + +interface PendingInteractionToolCall { + id: string; + name: string; + argumentsText: string; + argumentsObject: Record; +} + +type InteractionInputStep = + | InteractionUserInputStep + | InteractionModelOutputStep + | InteractionFunctionCallStep + | InteractionFunctionResultStep; + +type InteractionStep = + | InteractionModelOutputStep + | InteractionFunctionCallStep + | InteractionFunctionResultStep + | InteractionThoughtStep + | InteractionUserInputStep; + +type InteractionThinkingLevel = "minimal" | "low" | "medium" | "high"; + +interface InteractionGenerationConfig { + temperature?: number; + top_p?: number; + top_k?: number; + min_p?: number; + presence_penalty?: number; + frequency_penalty?: number; + repetition_penalty?: number; + max_output_tokens?: number; + thinking_level?: InteractionThinkingLevel; + thinking_budget?: number; +} + +interface GoogleInteractionRequest { + model: string; + input: InteractionInputStep[]; + stream: true; + previous_interaction_id?: string; + system_instruction?: string; + tools?: { functionDeclarations: Record[] }[]; + store?: boolean; + generation_config?: InteractionGenerationConfig; + service_tier?: string; +} + +interface InteractionUsage { + total_input_tokens?: number; + total_cached_tokens?: number; + total_output_tokens?: number; + total_thought_tokens?: number; + total_tokens?: number; +} + +interface InteractionResource { + id?: string; + status?: string; + usage?: InteractionUsage; +} + +interface InteractionStreamMetadata { + total_usage?: InteractionUsage; +} + +interface InteractionSseEvent { + event_type?: string; + index?: number; + step?: InteractionStep; + delta?: + | InteractionContent + | InteractionFunctionCallStep + | InteractionThoughtStep + | InteractionThoughtSummaryDelta + | InteractionThoughtSignatureDelta + | InteractionArgumentsDelta; + interaction?: InteractionResource; + interaction_id?: string; + status?: string; + metadata?: InteractionStreamMetadata; + error?: { message?: string; code?: string | number }; +} + +function emptyUsage(): Usage { + return { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; +} + +function getGoogleInteractionsState( + providerSessionState: Map | undefined, + create: boolean, +): GoogleInteractionsProviderSessionState | undefined { + if (!providerSessionState) return undefined; + const existing = providerSessionState.get(GOOGLE_INTERACTIONS_STATE_KEY) as + | GoogleInteractionsProviderSessionState + | undefined; + if (existing || !create) return existing; + const state: GoogleInteractionsProviderSessionState = { close: () => {} }; + providerSessionState.set(GOOGLE_INTERACTIONS_STATE_KEY, state); + return state; +} + +function interactionContentFromText(text: string): InteractionContent[] { + return text.length === 0 ? [] : [{ type: "text", text }]; +} + +function interactionContentFromParts(parts: readonly (TextContent | ImageContent)[]): InteractionContent[] { + const content: InteractionContent[] = []; + for (const part of parts) { + if (part.type === "text") { + if (part.text.length > 0) content.push({ type: "text", text: part.text }); + } else { + content.push({ type: "image", data: part.data, mime_type: part.mimeType }); + } + } + return content; +} + +function userInputStepFromMessage(message: Extract): InteractionUserInputStep { + const content = + typeof message.content === "string" + ? interactionContentFromText(message.content) + : interactionContentFromParts(message.content); + return { type: "user_input", content }; +} + +function functionResultStepFromMessage( + message: Extract, +): InteractionFunctionResultStep { + const result = interactionContentFromParts(message.content); + return { + type: "function_result", + name: message.toolName, + call_id: message.toolCallId, + result: result.length > 0 ? result : [{ type: "text", text: "" }], + ...(message.isError ? { is_error: true } : {}), + }; +} + +function appendAssistantInteractionSteps(message: AssistantMessage, steps: InteractionInputStep[]): void { + let modelContent: InteractionContent[] = []; + const flushModelContent = (): void => { + if (modelContent.length === 0) return; + steps.push({ type: "model_output", content: modelContent }); + modelContent = []; + }; + + for (const block of message.content) { + if (block.type === "text") { + if (block.text.length > 0) modelContent.push({ type: "text", text: block.text }); + } else if (block.type === "toolCall") { + flushModelContent(); + steps.push({ type: "function_call", id: block.id, name: block.name, arguments: block.arguments }); + } + } + flushModelContent(); +} + +function interactionMessagesAfterAnchor( + messages: readonly Message[], + anchorIndex: number | undefined, +): readonly Message[] { + return anchorIndex === undefined ? messages : messages.slice(anchorIndex + 1); +} + +function buildInteractionInput(context: Context, anchorIndex: number | undefined): InteractionInputStep[] { + const input: InteractionInputStep[] = []; + for (const message of interactionMessagesAfterAnchor(context.messages, anchorIndex)) { + if (message.role === "user" || message.role === "developer") { + const step = userInputStepFromMessage(message); + if (step.content.length > 0) input.push(step); + } else if (message.role === "toolResult") { + input.push(functionResultStepFromMessage(message)); + } else if (anchorIndex === undefined) { + appendAssistantInteractionSteps(message, input); + } + } + return input.length > 0 ? input : [{ type: "user_input", content: [{ type: "text", text: "" }] }]; +} + +function toInteractionThinkingLevel(level: GoogleThinkingLevel): InteractionThinkingLevel | undefined { + switch (level) { + case "MINIMAL": + return "minimal"; + case "LOW": + return "low"; + case "MEDIUM": + return "medium"; + case "HIGH": + return "high"; + case "THINKING_LEVEL_UNSPECIFIED": + return undefined; + } +} + +function buildInteractionGenerationConfig(options: GoogleOptions | undefined): InteractionGenerationConfig | undefined { + const config: InteractionGenerationConfig = {}; + if (options?.temperature !== undefined) config.temperature = options.temperature; + if (options?.topP !== undefined) config.top_p = options.topP; + if (options?.topK !== undefined) config.top_k = options.topK; + if (options?.minP !== undefined) config.min_p = options.minP; + if (options?.presencePenalty !== undefined) config.presence_penalty = options.presencePenalty; + if (options?.frequencyPenalty !== undefined) config.frequency_penalty = options.frequencyPenalty; + if (options?.repetitionPenalty !== undefined) config.repetition_penalty = options.repetitionPenalty; + if (options?.maxTokens !== undefined) config.max_output_tokens = options.maxTokens; + if (options?.thinking?.level !== undefined) { + const thinkingLevel = toInteractionThinkingLevel(options.thinking.level); + if (thinkingLevel !== undefined) config.thinking_level = thinkingLevel; + } else if (options?.thinking?.budgetTokens !== undefined) { + config.thinking_budget = options.thinking.budgetTokens; + } + return Object.keys(config).length > 0 ? config : undefined; +} + +function buildInteractionRequest( + model: GoogleInteractionsModel, + context: Context, + options: GoogleOptions | undefined, + anchor: InteractionAnchor, +): GoogleInteractionRequest { + const systemInstruction = normalizeSystemPrompts(context.systemPrompt).join("\n\n"); + const generationConfig = buildInteractionGenerationConfig(options); + return { + model: model.id, + input: buildInteractionInput(context, anchor.messageIndex), + stream: true, + ...(anchor.id !== undefined ? { previous_interaction_id: anchor.id } : {}), + ...(systemInstruction.length > 0 ? { system_instruction: systemInstruction } : {}), + ...(context.tools && context.tools.length > 0 ? { tools: convertTools(context.tools, model) } : {}), + ...(options?.storeInteraction !== undefined ? { store: options.storeInteraction } : {}), + ...(generationConfig !== undefined ? { generation_config: generationConfig } : {}), + ...(shouldSendServiceTier(options?.serviceTier, model.provider) ? { service_tier: options?.serviceTier } : {}), + }; +} + +function applyInteractionUsage( + model: GoogleInteractionsModel, + output: AssistantMessage, + usage: InteractionUsage, +): void { + const thinkingTokens = usage.total_thought_tokens ?? 0; + output.usage = { + input: (usage.total_input_tokens ?? 0) - (usage.total_cached_tokens ?? 0), + output: (usage.total_output_tokens ?? 0) + thinkingTokens, + cacheRead: usage.total_cached_tokens ?? 0, + cacheWrite: 0, + totalTokens: usage.total_tokens ?? 0, + ...(thinkingTokens > 0 ? { reasoningTokens: thinkingTokens } : {}), + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; + calculateCost(model, output.usage); +} + +function parseInteractionFunctionCall(value: unknown): InteractionFunctionCallStep | undefined { + if (!value || typeof value !== "object" || Array.isArray(value)) return undefined; + const record = value as Record; + if (record.type !== "function_call") return undefined; + if (typeof record.id !== "string" || typeof record.name !== "string") return undefined; + const args = record.arguments; + return { + type: "function_call", + id: record.id, + name: record.name, + arguments: args && typeof args === "object" && !Array.isArray(args) ? { ...args } : {}, + }; +} + +function pendingToolCallFromStep(call: InteractionFunctionCallStep): PendingInteractionToolCall { + return { + id: call.id, + name: call.name, + argumentsText: "", + argumentsObject: call.arguments, + }; +} + +function parseInteractionArguments(text: string): Record { + if (text.trim().length === 0) return {}; + try { + const parsed: unknown = JSON.parse(text); + if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return { ...parsed }; + } catch { + return {}; + } + return {}; +} + +/** Provider-specific URL, headers, and fetch implementation for an Interactions request. */ +export interface GoogleInteractionsPlan { + url: string; + headers: Record; + fetch?: FetchImpl; +} + +/** + * Streams Gemini Interactions API model-mode responses for direct Google and Vertex providers. + * + * `fallback`, when supplied, is the legacy `:streamGenerateContent` stream factory. It runs + * transparently — forwarding its events into this stream — when the Interactions attempt fails + * before any content is emitted with a signal that the model/endpoint does not support + * Interactions (HTTP 404/400). Provide it only for auto-selected Interactions requests so an + * explicit `useInteractionsApi: true` still surfaces failures. + */ +export function streamGoogleInteractions(args: { + model: Model; + context: Context; + options: GoogleSharedStreamOptions | undefined; + api: T; + anchor: InteractionAnchor; + state: GoogleInteractionsProviderSessionState | undefined; + prepare: () => GoogleInteractionsPlan | Promise; + fallback?: () => AssistantMessageEventStream; +}): AssistantMessageEventStream { + const { model, context, options, anchor, state } = args; + const stream = new AssistantMessageEventStream(); + const output: AssistantMessage = { + role: "assistant", + content: [], + api: args.api, + provider: model.provider, + model: model.id, + usage: emptyUsage(), + stopReason: "stop", + timestamp: Date.now(), + }; + const storeInteraction = options?.storeInteraction !== false; + + void (async () => { + let started = false; + let sawTerminal = false; + let currentTextBlock: TextContent | undefined; + let currentThinkingBlock: Extract | undefined; + let pendingThinkingSignature: string | undefined; + const stepKinds = new Map(); + const pendingToolCalls = new Map(); + const ensureStarted = (): void => { + if (started) return; + stream.push({ type: "start", partial: output }); + started = true; + }; + const endOpenBlocks = (): void => { + if (currentTextBlock) { + stream.push({ + type: "text_end", + contentIndex: output.content.indexOf(currentTextBlock), + content: currentTextBlock.text, + partial: output, + }); + currentTextBlock = undefined; + } + if (currentThinkingBlock) { + stream.push({ + type: "thinking_end", + contentIndex: output.content.indexOf(currentThinkingBlock), + content: currentThinkingBlock.thinking, + partial: output, + }); + currentThinkingBlock = undefined; + } + }; + const emitText = (text: string): void => { + if (text.length === 0) return; + ensureStarted(); + if (!currentTextBlock) { + if (currentThinkingBlock) endOpenBlocks(); + currentTextBlock = { type: "text", text: "" }; + output.content.push(currentTextBlock); + stream.push({ type: "text_start", contentIndex: output.content.length - 1, partial: output }); + } + currentTextBlock.text += text; + stream.push({ + type: "text_delta", + contentIndex: output.content.indexOf(currentTextBlock), + delta: text, + partial: output, + }); + }; + const applyThinkingSignature = (signature: string | undefined): void => { + if (!signature) return; + if (currentThinkingBlock) { + currentThinkingBlock.thinkingSignature = signature; + } else { + pendingThinkingSignature = signature; + } + }; + const emitThinking = (text: string): void => { + if (text.length === 0) return; + ensureStarted(); + if (!currentThinkingBlock) { + if (currentTextBlock) endOpenBlocks(); + currentThinkingBlock = { type: "thinking", thinking: "", thinkingSignature: pendingThinkingSignature }; + pendingThinkingSignature = undefined; + output.content.push(currentThinkingBlock); + stream.push({ type: "thinking_start", contentIndex: output.content.length - 1, partial: output }); + } + currentThinkingBlock.thinking += text; + stream.push({ + type: "thinking_delta", + contentIndex: output.content.indexOf(currentThinkingBlock), + delta: text, + partial: output, + }); + }; + const emitToolCall = (call: InteractionFunctionCallStep): void => { + ensureStarted(); + endOpenBlocks(); + const toolCall: ToolCall = { + type: "toolCall", + id: call.id, + name: call.name, + arguments: call.arguments, + }; + output.content.push(toolCall); + const contentIndex = output.content.length - 1; + stream.push({ type: "toolcall_start", contentIndex, partial: output }); + stream.push({ + type: "toolcall_delta", + contentIndex, + delta: JSON.stringify(toolCall.arguments), + partial: output, + }); + stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); + }; + const emitPendingToolCall = (pending: PendingInteractionToolCall): void => { + emitToolCall({ + type: "function_call", + id: pending.id, + name: pending.name, + arguments: + pending.argumentsText.length > 0 + ? parseInteractionArguments(pending.argumentsText) + : pending.argumentsObject, + }); + }; + + let prepared = false; + try { + const plan = await args.prepare(); + prepared = true; + let requestBody: unknown = buildInteractionRequest(model, context, options, anchor); + const replacement = await options?.onPayload?.(requestBody, model); + if (replacement !== undefined) requestBody = replacement; + const response = await fetchWithRetry(() => plan.url, { + method: "POST", + headers: { + ...plan.headers, + "Content-Type": "application/json", + Accept: "text/event-stream", + }, + body: JSON.stringify(requestBody), + signal: options?.signal, + fetch: plan.fetch, + }); + if (!response.ok) { + const errorText = await response.text().catch(() => ""); + throw new AIError.GoogleApiError( + `Google Interactions API error (${response.status}): ${errorText}`, + response.status, + { + headers: response.headers, + }, + ); + } + if (!response.body) { + throw new AIError.ProviderResponseError("Google Interactions API returned an empty response body", { + provider: model.provider, + kind: "empty-body", + }); + } + for await (const event of readSseJson(response.body, options?.signal, sse => + options?.onSseEvent?.({ event: sse.event, data: sse.data, raw: [...sse.raw] }, model), + )) { + if (event.error) { + throw new AIError.ProviderResponseError(event.error.message ?? "Google Interactions API stream error", { + provider: model.provider, + kind: "runtime", + }); + } + if (event.metadata?.total_usage) applyInteractionUsage(model, output, event.metadata.total_usage); + if (event.event_type === "interaction.created") { + if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; + } else if (event.event_type === "step.start" && event.index !== undefined && event.step) { + stepKinds.set(event.index, event.step.type); + const call = parseInteractionFunctionCall(event.step); + if (call) { + pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); + } else if (event.step.type === "thought") { + applyThinkingSignature(event.step.signature); + for (const item of event.step.summary ?? []) { + if (item.type === "text") emitThinking(item.text); + } + } + } else if (event.event_type === "step.delta" && event.index !== undefined && event.delta) { + const call = parseInteractionFunctionCall(event.delta); + if (call) { + pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); + } else if (event.delta.type === "text") { + if (stepKinds.get(event.index) === "thought") emitThinking(event.delta.text); + else emitText(event.delta.text); + } else if (event.delta.type === "thought_summary") { + if (event.delta.content?.type === "text") emitThinking(event.delta.content.text); + } else if (event.delta.type === "thought_signature") { + applyThinkingSignature(event.delta.signature); + } else if (event.delta.type === "arguments_delta") { + const pending = pendingToolCalls.get(event.index); + if (pending && event.delta.arguments) pending.argumentsText += event.delta.arguments; + } + } else if (event.event_type === "step.stop" && event.index !== undefined) { + const stepKind = stepKinds.get(event.index); + const pending = pendingToolCalls.get(event.index); + if (pending) { + emitPendingToolCall(pending); + pendingToolCalls.delete(event.index); + } else { + endOpenBlocks(); + if (stepKind === "thought") pendingThinkingSignature = undefined; + } + stepKinds.delete(event.index); + } else if (event.event_type === "interaction.completed" || event.event_type === "interaction.complete") { + if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; + if (event.interaction?.usage) applyInteractionUsage(model, output, event.interaction.usage); + for (const pending of pendingToolCalls.values()) emitPendingToolCall(pending); + pendingToolCalls.clear(); + endOpenBlocks(); + output.stopReason = + event.interaction?.status === "requires_action" || + output.content.some(block => block.type === "toolCall") + ? "toolUse" + : "stop"; + if (storeInteraction && state) state.lastInteractionId = output.responseId; + sawTerminal = true; + ensureStarted(); + stream.push({ type: "done", reason: output.stopReason, message: output }); + } + } + if (!sawTerminal) { + throw new AIError.ProviderResponseError("Google Interactions API stream ended without a terminal event", { + provider: model.provider, + kind: "incomplete-stream", + }); + } + } catch (error) { + // Auto-selected Interactions degrades to `:streamGenerateContent` when no content has + // streamed yet and the failure means Interactions can't serve this request — the bearer + // credential couldn't be resolved (prepare threw) or the model/endpoint rejected it + // (HTTP 404/400). Provider 401/403/429/5xx still surface. Mirrors the OpenAI Responses + // `previous_response_id` fallback. + const unsupported = error instanceof AIError.GoogleApiError && (error.status === 404 || error.status === 400); + if (!started && args.fallback && !options?.signal?.aborted && (!prepared || unsupported)) { + for await (const event of args.fallback()) stream.push(event); + return; + } + output.stopReason = options?.signal?.aborted ? "aborted" : "error"; + output.errorMessage = error instanceof Error ? error.message : String(error); + stream.push({ type: "error", reason: output.stopReason, error: output }); + } + })(); + + return stream; +} + +function findAssistantInteractionAnchor( + context: Context, + interactionId: string, + provider: string, +): InteractionAnchor | undefined { + for (let index = context.messages.length - 1; index >= 0; index -= 1) { + const message = context.messages[index]; + if (message?.role === "assistant" && message.provider === provider && message.responseId === interactionId) { + return { id: interactionId, messageIndex: index }; + } + } + return undefined; +} + +function latestAssistantInteractionAnchor(context: Context, provider: string): InteractionAnchor | undefined { + for (let index = context.messages.length - 1; index >= 0; index -= 1) { + const message = context.messages[index]; + if (message?.role === "assistant" && message.provider === provider && message.responseId) { + return { id: message.responseId, messageIndex: index }; + } + } + return undefined; +} + +function resolveInteractionAnchor( + context: Context, + explicitPreviousInteractionId: string | undefined, + state: GoogleInteractionsProviderSessionState | undefined, + provider: string, +): InteractionAnchor { + if (explicitPreviousInteractionId !== undefined) { + return ( + findAssistantInteractionAnchor(context, explicitPreviousInteractionId, provider) ?? { + id: explicitPreviousInteractionId, + } + ); + } + const lineageAnchor = latestAssistantInteractionAnchor(context, provider); + if (lineageAnchor) return lineageAnchor; + if (state?.lastInteractionId) + return findAssistantInteractionAnchor(context, state.lastInteractionId, provider) ?? {}; + return {}; +} + +/** + * Whether a model is served by the Gemini Interactions API. Interactions is a Gemini 3-era + * transport, so the catalog subset that supports it is Gemini 3.0+. Older Gemini and non-Gemini + * ids keep `:streamGenerateContent`, which covers the full catalog. + */ +export function modelSupportsInteractions(model: Pick): boolean { + const parsed = parseGeminiModel(model.id); + return parsed !== null && parsed.version.major >= 3; +} + +/** + * Resolves whether a Google provider call should use Interactions and which lineage anchor to send. + * + * Precedence: explicit `useInteractionsApi: false` always wins (force generateContent); otherwise + * Interactions engages when explicitly requested, when continuing a stored interaction + * (`previousInteractionId`/assistant lineage/session state), or when `autoEligible` (the + * zero-config default for the capable model subset on the official endpoint). `auto` flags the + * last case for the caller — it is the only mode that wires up the generateContent fallback. + */ +export function resolveInteractionDispatch(args: { + context: Context; + options: GoogleSharedStreamOptions | undefined; + provider: string; + autoEligible: boolean; +}): { + useInteractions: boolean; + auto: boolean; + anchor: InteractionAnchor; + state: GoogleInteractionsProviderSessionState | undefined; +} { + const explicitPreviousInteractionId = args.options?.previousInteractionId; + if (args.options?.storeInteraction === false && explicitPreviousInteractionId !== undefined) { + throw new AIError.ConfigurationError( + "Google Interactions API cannot combine storeInteraction:false with previousInteractionId.", + ); + } + const explicitOptOut = args.options?.useInteractionsApi === false; + const explicitOptIn = args.options?.useInteractionsApi === true || explicitPreviousInteractionId !== undefined; + const storageEnabled = args.options?.storeInteraction !== false; + const existingState = storageEnabled + ? getGoogleInteractionsState(args.options?.providerSessionState, false) + : undefined; + const anchor = explicitOptOut + ? {} + : resolveInteractionAnchor(args.context, explicitPreviousInteractionId, existingState, args.provider); + const useInteractions = !explicitOptOut && (explicitOptIn || anchor.id !== undefined || args.autoEligible); + const auto = useInteractions && !explicitOptIn; + const interactionState = + useInteractions && storageEnabled + ? getGoogleInteractionsState( + args.options?.providerSessionState, + args.options?.providerSessionState !== undefined, + ) + : undefined; + return { useInteractions, auto, anchor, state: interactionState }; +} diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index ae5cd9d0b..8f9b70cad 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -77,6 +77,19 @@ export interface GoogleSharedStreamOptions extends StreamOptions { }; /** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */ serviceTier?: ServiceTier; + /** + * Continues a Gemini Interactions API conversation from a stored interaction. + * When set on the direct Google provider, the request uses `/interactions` + * with `previous_interaction_id` instead of the legacy generateContent stream. + */ + previousInteractionId?: string; + /** + * Uses the Gemini Interactions API for direct Google requests, storing the + * returned interaction id on the assistant response for follow-up turns. + */ + useInteractionsApi?: boolean; + /** Overrides Interactions API request storage; default is the API default (`true`). */ + storeInteraction?: boolean; } /** diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index 878286a53..66b1868b9 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -2,7 +2,13 @@ import { $env } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; -import { getVertexAccessToken } from "./google-auth"; +import { getVertexAccessToken, hasVertexBearerCredentialsHint } from "./google-auth"; +import { + type GoogleInteractionsPlan, + modelSupportsInteractions, + resolveInteractionDispatch, + streamGoogleInteractions, +} from "./google-interactions"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -16,87 +22,128 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions { } const API_VERSION = "v1"; +const INTERACTIONS_API_VERSION = "v1beta1"; +const INTERACTIONS_API_REVISION = "2026-05-20"; export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( model: Model<"google-vertex">, context: Context, options?: GoogleVertexOptions, -): AssistantMessageEventStream => - streamGoogleGenAI({ - model, - options, - api: "google-vertex", - retainTextSignature: true, - prepare: async (): Promise => { - const apiKey = resolveApiKey(options); - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - params.config ||= {}; - if (!params.config.safetySettings) { - params.config.safetySettings = [ - { - category: "HARM_CATEGORY_HATE_SPEECH", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_DANGEROUS_CONTENT", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_HARASSMENT", - threshold: "OFF", - }, - ]; - } - const baseHeaders: Record = { - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - }; - // Vertex AI ignores a `serviceTier` request-body field (unlike the direct - // Gemini API); priority must travel as a request header. Only `priority` - // has a documented Vertex request control — `flex` has none, so it's a no-op. - if (options?.serviceTier === "priority") { - baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; - } +): AssistantMessageEventStream => { + const runGenerateContent = (): AssistantMessageEventStream => + streamGoogleGenAI({ + model, + options, + api: "google-vertex", + retainTextSignature: true, + prepare: async (): Promise => { + const apiKey = resolveApiKey(options); + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + params.config ||= {}; + if (!params.config.safetySettings) { + params.config.safetySettings = [ + { + category: "HARM_CATEGORY_HATE_SPEECH", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_DANGEROUS_CONTENT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_HARASSMENT", + threshold: "OFF", + }, + ]; + } + const baseHeaders: Record = { + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + }; + // Vertex AI ignores a `serviceTier` request-body field (unlike the direct + // Gemini API); priority must travel as a request header. Only `priority` + // has a documented Vertex request control — `flex` has none, so it's a no-op. + if (options?.serviceTier === "priority") { + baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; + } - if (apiKey) { - // Explicit `location` is a deliberate residency choice: honor it and let - // a 404 surface. An ambient env-derived region falls back to the global - // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working - // global-only request. - const explicitLocation = options?.location; - const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; + if (apiKey) { + // Explicit `location` is a deliberate residency choice: honor it and let + // a 404 surface. An ambient env-derived region falls back to the global + // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working + // global-only request. + const explicitLocation = options?.location; + const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; + const host = resolveEndpointHost(location); + const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; + return { + params, + url: `https://${host}/${path}`, + fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, + headers: { + ...baseHeaders, + "x-goog-api-key": apiKey, + }, + fetch: options?.fetch, + }; + } + + const project = resolveProject(options); + const location = resolveLocation(options); + const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); const host = resolveEndpointHost(location); - const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; - const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; + const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { params, - url: `https://${host}/${path}`, - fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, - headers: { - ...baseHeaders, - "x-goog-api-key": apiKey, - }, + url, + headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, fetch: options?.fetch, }; - } + }, + }); + // Default Gemini 3+ onto Interactions whenever a bearer credential source exists (ADC file, + // `GOOGLE_APPLICATION_CREDENTIALS`, or an explicit access-token env). Interactions needs bearer + // auth, so express API-key-only setups stay on generateContent — and an express key, when + // present, still serves the generateContent fallback. Interactions always targets the official + // global `aiplatform` host; the fallback also recovers ids the endpoint rejects. + const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ + context, + options, + provider: model.provider, + autoEligible: modelSupportsInteractions(model) && hasVertexBearerCredentialsHint(), + }); + if (!useInteractions) return runGenerateContent(); + + return streamGoogleInteractions({ + model, + context, + options, + api: "google-vertex", + anchor, + state, + prepare: async (): Promise => { const project = resolveProject(options); - const location = resolveLocation(options); const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); - const host = resolveEndpointHost(location); - const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { - params, - url, - headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, + url: `https://aiplatform.googleapis.com/${INTERACTIONS_API_VERSION}/projects/${project}/locations/global/interactions`, + headers: { + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + Authorization: `Bearer ${accessToken}`, + "Api-Revision": INTERACTIONS_API_REVISION, + }, fetch: options?.fetch, }; }, + fallback: auto ? runGenerateContent : undefined, }); +}; function resolveApiKey(options?: GoogleVertexOptions): string | undefined { // options.apiKey may contain sentinel values like "" or "N/A" diff --git a/packages/ai/src/providers/google.ts b/packages/ai/src/providers/google.ts index 24b5a1280..3bd8ed7ff 100644 --- a/packages/ai/src/providers/google.ts +++ b/packages/ai/src/providers/google.ts @@ -2,6 +2,7 @@ import * as AIError from "../error"; import { getEnvApiKey } from "../stream"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; +import { modelSupportsInteractions, resolveInteractionDispatch, streamGoogleInteractions } from "./google-interactions"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -17,29 +18,70 @@ export const streamGoogle: StreamFunction<"google-generative-ai"> = ( model: Model<"google-generative-ai">, context: Context, options?: GoogleOptions, -): AssistantMessageEventStream => - streamGoogleGenAI({ +): AssistantMessageEventStream => { + const apiKey = options?.apiKey || getEnvApiKey(model.provider); + if (!apiKey) { + throw new AIError.MissingApiKeyError( + undefined, + "Google Generative AI requires an API key (GEMINI_API_KEY or options.apiKey).", + ); + } + + const runGenerateContent = (): AssistantMessageEventStream => + streamGoogleGenAI({ + model, + options, + api: "google-generative-ai", + prepare: (): GoogleGenAIRequestPlan => { + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + // `model.baseUrl` already includes the API version segment when set (mirrors the + // `apiVersion: ""` reset that the SDK relied on for custom base URLs). + const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; + const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; + const headers: Record = { + "x-goog-api-key": apiKey, + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + }; + return { params, url, headers, fetch: options?.fetch }; + }, + }); + + // Default Gemini 3+ on the official endpoint onto Interactions (custom proxy base URLs keep + // generateContent, which serves the full catalog). The fallback recovers ids the endpoint rejects. + const trimmedBase = model.baseUrl?.trim(); + let officialEndpoint = !trimmedBase; + if (trimmedBase) { + try { + officialEndpoint = new URL(trimmedBase).hostname === "generativelanguage.googleapis.com"; + } catch { + officialEndpoint = false; + } + } + const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ + context, + options, + provider: model.provider, + autoEligible: officialEndpoint && modelSupportsInteractions(model), + }); + if (!useInteractions) return runGenerateContent(); + + return streamGoogleInteractions({ model, + context, options, api: "google-generative-ai", - prepare: (): GoogleGenAIRequestPlan => { - const apiKey = options?.apiKey || getEnvApiKey(model.provider); - if (!apiKey) { - throw new AIError.MissingApiKeyError( - undefined, - "Google Generative AI requires an API key (GEMINI_API_KEY or options.apiKey).", - ); - } - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - // `model.baseUrl` already includes the API version segment when set (mirrors the - // `apiVersion: ""` reset that the SDK relied on for custom base URLs). - const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; - const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; - const headers: Record = { + anchor, + state, + prepare: () => ({ + url: `${trimmedBase || DEFAULT_GENERATIVE_LANGUAGE_BASE}/interactions`, + headers: { "x-goog-api-key": apiKey, ...(model.headers ?? {}), ...(options?.headers ?? {}), - }; - return { params, url, headers, fetch: options?.fetch }; - }, + }, + fetch: options?.fetch, + }), + fallback: auto ? runGenerateContent : undefined, }); +}; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 68f1410c9..102a7c717 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1355,6 +1355,9 @@ function mapOptionsForApi( streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs, streamIdleTimeoutMs: options?.streamIdleTimeoutMs, providerSessionState: options?.providerSessionState, + useInteractionsApi: options?.useInteractionsApi, + storeInteraction: options?.storeInteraction, + previousInteractionId: options?.previousInteractionId, maxInFlightRequests: options?.maxInFlightRequests, onPayload: options?.onPayload, onResponse: options?.onResponse, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index bc1bb0f72..b6c0467df 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -374,6 +374,22 @@ export interface StreamOptions { * Providers can use this to persist transport/session state between turns. */ providerSessionState?: Map; + /** + * Force Gemini model-mode Interactions API transport for providers that support it. + * When unset, those providers may still use Interactions to continue known + * server-side conversation lineage via `previousInteractionId` or stored state. + */ + useInteractionsApi?: boolean; + /** + * Whether supported Interactions transports should store server-side conversation + * state and return response ids for follow-up turns. Defaults to true. + */ + storeInteraction?: boolean; + /** + * Explicit Interactions response id to continue. Mutually exclusive with + * `storeInteraction: false` because the follow-up itself must be storable. + */ + previousInteractionId?: string; /** * Optional per-provider concurrent request cap for LLM stream calls. Keys are * provider ids (`model.provider`); positive numeric values cap in-flight diff --git a/packages/ai/test/google-empty-response-retry.test.ts b/packages/ai/test/google-empty-response-retry.test.ts index 005c6f8ad..b1e4db870 100644 --- a/packages/ai/test/google-empty-response-retry.test.ts +++ b/packages/ai/test/google-empty-response-retry.test.ts @@ -89,7 +89,8 @@ describe("Google empty-response retry (public + Vertex path)", () => { return calls === 1 ? sse(genaiChunk("")) : sse(genaiChunk("Hello!")); }; - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); + // Pin the generateContent transport: gemini-3 ids now auto-route to Interactions by default. + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); const { events, starts } = await drain(stream); const result = await stream.result(); @@ -107,7 +108,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { return sse(genaiChunk("")); }; - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); const result = await stream.result(); expect(calls).toBe(3); // MAX_EMPTY_STREAM_RETRIES (2) + 1 initial attempt @@ -137,6 +138,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, + useInteractionsApi: false, }); const { events } = await drain(stream); const result = await stream.result(); @@ -194,6 +196,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, + useInteractionsApi: false, }); const result = await stream.result(); diff --git a/packages/ai/test/google-interactions.test.ts b/packages/ai/test/google-interactions.test.ts new file mode 100644 index 000000000..136fe2d95 --- /dev/null +++ b/packages/ai/test/google-interactions.test.ts @@ -0,0 +1,511 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; +import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth"; +import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; +import { streamSimple } from "@oh-my-pi/pi-ai/stream"; +import type { AssistantMessage, Context, FetchImpl, Model, Tool, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +function googleModel(baseUrl = "https://generativelanguage.googleapis.com/v1beta"): Model<"google-generative-ai"> { + return buildModel({ + id: "gemini-3.5-flash", + name: "Gemini 3.5 Flash", + api: "google-generative-ai", + provider: "google", + baseUrl, + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8_192, + }); +} + +function vertexModel(id = "gemini-3.5-flash"): Model<"google-vertex"> { + return buildModel({ + id, + name: id, + api: "google-vertex", + provider: "google-vertex", + baseUrl: "", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8_192, + }); +} + +function sseResponse(events: readonly unknown[]): Response { + const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; + return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } }); +} + +const weatherTool: Tool = { + name: "get_weather", + description: "Get weather", + parameters: { + type: "object", + properties: { city: { type: "string" } }, + required: ["city"], + additionalProperties: false, + }, +}; + +describe("Google Interactions API", () => { + it("chains tool results with previous_interaction_id from the prior assistant response", async () => { + const model = googleModel(); + const requestBodies: unknown[] = []; + let calls = 0; + const fetchMock: FetchImpl = async (_input, init) => { + requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); + calls += 1; + if (calls === 1) { + return sseResponse([ + { + event_type: "interaction.created", + interaction: { id: "int_1", status: "in_progress" }, + }, + { event_type: "step.start", index: 0, step: { type: "thought" } }, + { + event_type: "step.delta", + index: 0, + delta: { type: "thought_signature", signature: "thought_sig_1" }, + }, + { + event_type: "step.delta", + index: 0, + delta: { type: "thought_summary", content: { type: "text", text: "Checking weather.\n" } }, + }, + { event_type: "step.stop", index: 0 }, + { + event_type: "step.start", + index: 1, + step: { + type: "function_call", + id: "call_weather", + name: "get_weather", + arguments: {}, + }, + }, + { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: '{"city":"Bos' } }, + { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: 'ton"}' } }, + { event_type: "step.stop", index: 1 }, + { + event_type: "interaction.completed", + interaction: { + id: "int_1", + status: "requires_action", + usage: { total_input_tokens: 10, total_output_tokens: 2, total_tokens: 12 }, + }, + }, + ]); + } + return sseResponse([ + { event_type: "interaction.created", interaction: { id: "int_2", status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "model_output" } }, + { event_type: "step.delta", index: 0, delta: { type: "text", text: "Sunny." } }, + { event_type: "step.stop", index: 0 }, + { + event_type: "interaction.completed", + interaction: { + id: "int_2", + status: "completed", + usage: { total_input_tokens: 3, total_output_tokens: 1, total_tokens: 4 }, + }, + }, + ]); + }; + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + + const firstContext: Context = { + systemPrompt: ["Use concise weather reports."], + messages: [{ role: "user", content: "Need weather", timestamp: 1 }], + tools: [weatherTool], + }; + const first = await streamGoogle(model, firstContext, { + apiKey: "test-key", + fetch: fetchMock, + useInteractionsApi: true, + thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, + }).result(); + + expect(first.responseId).toBe("int_1"); + expect(first.stopReason).toBe("toolUse"); + expect(first.content).toEqual([ + { type: "thinking", thinking: "Checking weather.\n", thinkingSignature: "thought_sig_1" }, + { type: "toolCall", id: "call_weather", name: "get_weather", arguments: { city: "Boston" } }, + ]); + expect(requestBodies[0]).toMatchObject({ + model: "gemini-3.5-flash", + stream: true, + input: [{ type: "user_input", content: [{ type: "text", text: "Need weather" }] }], + system_instruction: "Use concise weather reports.", + tools: [{ functionDeclarations: [{ name: "get_weather" }] }], + generation_config: { thinking_level: "high" }, + }); + expect(requestBodies[0]).not.toHaveProperty("previous_interaction_id"); + expect(JSON.stringify(requestBodies[0])).not.toContain("thinking_budget"); + + const secondContext: Context = { + messages: [ + { role: "user", content: "Need weather", timestamp: 1 }, + first, + { + role: "toolResult", + toolCallId: "call_weather", + toolName: "get_weather", + content: [{ type: "text", text: "72F and sunny" }], + isError: false, + timestamp: 2, + }, + ], + tools: [weatherTool], + systemPrompt: ["Use concise weather reports."], + }; + const second = await streamGoogle(model, secondContext, { + apiKey: "test-key", + fetch: fetchMock, + thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, + }).result(); + + expect(second.responseId).toBe("int_2"); + expect(second.content).toEqual([{ type: "text", text: "Sunny." }]); + expect(requestBodies[1]).toMatchObject({ + previous_interaction_id: "int_1", + input: [ + { + type: "function_result", + name: "get_weather", + call_id: "call_weather", + result: [{ type: "text", text: "72F and sunny" }], + }, + ], + tools: [{ functionDeclarations: [{ name: "get_weather" }] }], + system_instruction: "Use concise weather reports.", + generation_config: { thinking_level: "high" }, + }); + expect(JSON.stringify(requestBodies[1])).not.toContain("Need weather"); + expect(JSON.stringify(requestBodies[1])).not.toContain("thinking_budget"); + }); + + it("does not expose or reuse interaction ids when storage is disabled", async () => { + const model = googleModel(); + const requestBodies: unknown[] = []; + const fetchMock: FetchImpl = async (_input, init) => { + requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); + return sseResponse([ + { event_type: "interaction.created", interaction: { id: "unstored_int", status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "model_output" } }, + { event_type: "step.delta", index: 0, delta: { type: "text", text: "Done." } }, + { event_type: "step.stop", index: 0 }, + { event_type: "interaction.completed", interaction: { id: "unstored_int", status: "completed" } }, + ]); + }; + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + + const result = await streamGoogle( + model, + { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, + { + apiKey: "test-key", + fetch: fetchMock, + useInteractionsApi: true, + storeInteraction: false, + }, + ).result(); + + expect(result.responseId).toBeUndefined(); + expect(requestBodies[0]).toMatchObject({ store: false }); + expect(() => + streamGoogle( + model, + { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, + { + apiKey: "test-key", + fetch: fetchMock, + storeInteraction: false, + previousInteractionId: "unstored_int", + }, + ), + ).toThrow(/storeInteraction:false/); + }); + + it("reads thought payload from step.start without leaking a prior signature", async () => { + const model = googleModel(); + const fetchMock: FetchImpl = async () => + sseResponse([ + { event_type: "interaction.created", interaction: { id: "int_3", status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "thought", signature: "stale_sig" } }, + { event_type: "step.stop", index: 0 }, + { + event_type: "step.start", + index: 1, + step: { type: "thought", summary: [{ type: "text", text: "Fresh plan.\n" }] }, + }, + { event_type: "step.stop", index: 1 }, + { event_type: "step.start", index: 2, step: { type: "model_output" } }, + { event_type: "step.delta", index: 2, delta: { type: "text", text: "Answer." } }, + { event_type: "step.stop", index: 2 }, + { event_type: "interaction.completed", interaction: { id: "int_3", status: "completed" } }, + ]); + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + + const result = await streamGoogle( + model, + { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, + { apiKey: "test-key", fetch: fetchMock, useInteractionsApi: true }, + ).result(); + + expect(result.content).toEqual([ + { type: "thinking", thinking: "Fresh plan.\n" }, + { type: "text", text: "Answer." }, + ]); + }); +}); + +function genaiSse(text: string): Response { + return new Response( + `data: ${JSON.stringify({ + candidates: [{ content: { parts: [{ text }] }, finishReason: "STOP" }], + usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 }, + })}\n\n`, + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); +} + +function interactionsTextSse( + id: string, + text: string, + terminal: "interaction.completed" | "interaction.complete" = "interaction.completed", +): Response { + return sseResponse([ + { event_type: "interaction.created", interaction: { id, status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "model_output" } }, + { event_type: "step.delta", index: 0, delta: { type: "text", text } }, + { event_type: "step.stop", index: 0 }, + { + event_type: terminal, + interaction: { + id, + status: "completed", + usage: { total_input_tokens: 10, total_output_tokens: 5, total_tokens: 15 }, + }, + }, + ]); +} + +interface CapturedCall { + url: string; + method: string; + headers: Headers; + body: unknown; +} + +function captureFetch(handler: (url: string) => Response): { fetch: FetchImpl; calls: CapturedCall[] } { + const calls: CapturedCall[] = []; + const fetchMock: FetchImpl = async (input, init) => { + const url = input instanceof Request ? input.url : String(input); + calls.push({ + url, + method: String(init?.method ?? "GET"), + headers: new Headers(init?.headers), + body: init?.body ? JSON.parse(String(init.body)) : undefined, + }); + return handler(url); + }; + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + return { fetch: fetchMock, calls }; +} + +const ZERO_USAGE: Usage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function assistantWithResponse( + api: "google-vertex" | "google-generative-ai", + provider: string, + responseId: string, +): AssistantMessage { + return { + role: "assistant", + api, + provider, + model: "gemini-3.5-flash", + content: [{ type: "text", text: "prev" }], + usage: ZERO_USAGE, + stopReason: "stop", + timestamp: 2, + responseId, + }; +} + +const userTurn: Context = { messages: [{ role: "user", content: "Hi there", timestamp: 1 }] }; + +describe("Google Interactions API — zero-config default + fallback", () => { + let savedToken: string | undefined; + beforeEach(() => { + // A bearer source makes the Vertex auto-gate fire deterministically on any machine, and + // `getVertexAccessToken` returns it directly (no OAuth/metadata round-trip in tests). + savedToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; + Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = "test-bearer"; + }); + afterEach(() => { + if (savedToken === undefined) delete Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; + else Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = savedToken; + __resetVertexTokenCache(); + }); + + it("auto-routes a capable Vertex model to Interactions under bearer auth", async () => { + const { fetch, calls } = captureFetch(() => interactionsTextSse("vint_1", "Hi")); + const result = await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + location: "us", + fetch, + }).result(); + + expect(calls).toHaveLength(1); + expect(calls[0].url).toBe("https://aiplatform.googleapis.com/v1beta1/projects/p/locations/global/interactions"); + expect(calls[0].method).toBe("POST"); + expect(calls[0].headers.get("Api-Revision")).toBe("2026-05-20"); + expect(calls[0].headers.get("Authorization")).toBe("Bearer test-bearer"); + expect(calls[0].body).toMatchObject({ + model: "gemini-3.5-flash", + stream: true, + input: [{ type: "user_input", content: [{ type: "text", text: "Hi there" }] }], + }); + expect(calls[0].body).not.toHaveProperty("contents"); + expect(calls[0].body).not.toHaveProperty("agent"); + expect(calls[0].body).not.toHaveProperty("environment"); + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "Hi" }]); + expect(result.responseId).toBe("vint_1"); + expect(result.usage.totalTokens).toBe(15); + }); + + it("keeps an older (sub-3) Vertex model on generateContent", async () => { + const { fetch, calls } = captureFetch(() => genaiSse("ok")); + await streamGoogleVertex(vertexModel("gemini-2.5-flash"), userTurn, { + project: "p", + location: "us", + fetch, + }).result(); + + expect(calls[0].url).toContain(":streamGenerateContent"); + expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); + }); + + it("honors useInteractionsApi:false on a capable Vertex model", async () => { + const { fetch, calls } = captureFetch(() => genaiSse("ok")); + await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + location: "us", + useInteractionsApi: false, + fetch, + }).result(); + + expect(calls[0].url).toContain(":streamGenerateContent"); + expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); + }); + + it("falls back to generateContent when auto Interactions is unsupported (404)", async () => { + const { fetch, calls } = captureFetch(url => + url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("recovered"), + ); + const result = await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + location: "us", + fetch, + }).result(); + + expect(calls[0].url).toContain("/interactions"); + expect(calls[1].url).toContain(":streamGenerateContent"); + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "recovered" }]); + }); + + it("surfaces the error (no fallback) when explicit Interactions is unsupported", async () => { + const { fetch, calls } = captureFetch(url => + url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("unexpected"), + ); + const result = await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + useInteractionsApi: true, + fetch, + }).result(); + + expect(result.stopReason).toBe("error"); + expect(calls.some(c => c.url.includes(":streamGenerateContent"))).toBe(false); + }); + + it("accepts interaction.complete as a terminal-event alias", async () => { + const { fetch } = captureFetch(() => interactionsTextSse("vint_2", "Done", "interaction.complete")); + const result = await streamGoogleVertex(vertexModel(), userTurn, { project: "p", fetch }).result(); + + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "Done" }]); + }); + + it("sends previous_interaction_id only for same-provider assistant lineage", async () => { + const sameProvider = captureFetch(() => interactionsTextSse("vint_3", "ok")); + await streamGoogleVertex( + vertexModel(), + { + messages: [ + { role: "user", content: "a", timestamp: 1 }, + assistantWithResponse("google-vertex", "google-vertex", "vint_prev"), + { role: "user", content: "b", timestamp: 3 }, + ], + }, + { project: "p", fetch: sameProvider.fetch }, + ).result(); + expect(sameProvider.calls[0].body).toMatchObject({ previous_interaction_id: "vint_prev" }); + + const wrongProvider = captureFetch(() => interactionsTextSse("vint_4", "ok")); + await streamGoogleVertex( + vertexModel(), + { + messages: [ + { role: "user", content: "a", timestamp: 1 }, + assistantWithResponse("google-generative-ai", "google", "gint_prev"), + { role: "user", content: "b", timestamp: 3 }, + ], + }, + { project: "p", fetch: wrongProvider.fetch }, + ).result(); + expect(wrongProvider.calls[0].body).not.toHaveProperty("previous_interaction_id"); + }); + + it("auto-routes a capable direct Google model on the official endpoint to Interactions", async () => { + const { fetch, calls } = captureFetch(() => interactionsTextSse("gint_1", "Hi")); + const result = await streamGoogle(googleModel(), userTurn, { apiKey: "k", fetch }).result(); + + expect(calls[0].url).toBe("https://generativelanguage.googleapis.com/v1beta/interactions"); + expect(calls[0].headers.get("x-goog-api-key")).toBe("k"); + expect(result.responseId).toBe("gint_1"); + }); + + it("keeps a custom-baseUrl direct Google model on generateContent", async () => { + const { fetch, calls } = captureFetch(() => genaiSse("ok")); + await streamGoogle(googleModel("https://proxy.example.com/v1beta"), userTurn, { + apiKey: "k", + fetch, + }).result(); + + expect(calls[0].url).toContain(":streamGenerateContent"); + expect(calls[0].url.startsWith("https://proxy.example.com/")).toBe(true); + expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); + }); + + it("threads the auto-default through streamSimple for a capable Google model", async () => { + const { fetch, calls } = captureFetch(() => interactionsTextSse("sint_1", "Hi")); + await streamSimple(googleModel(), userTurn, { apiKey: "k", fetch }).result(); + + expect(calls.some(c => c.url.includes("/interactions"))).toBe(true); + }); +}); diff --git a/packages/ai/test/google-service-tier.test.ts b/packages/ai/test/google-service-tier.test.ts index 15343cf84..ec541d847 100644 --- a/packages/ai/test/google-service-tier.test.ts +++ b/packages/ai/test/google-service-tier.test.ts @@ -75,7 +75,9 @@ const vertexModel: Model<"google-vertex"> = buildModel({ describe("Google service tier wire encoding", () => { it("Gemini API sends the tier in the request body, not a header", async () => { const { fetch, captured } = capturingFetch(); - await drain(streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch })); + await drain( + streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch, useInteractionsApi: false }), + ); const { headers, body } = captured(); expect(body.serviceTier).toBe("priority"); expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); @@ -99,7 +101,7 @@ describe("Google service tier wire encoding", () => { it("omits the tier entirely when unset", async () => { const { fetch, captured } = capturingFetch(); - await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch })); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch, useInteractionsApi: false })); expect(captured().body.serviceTier).toBeUndefined(); }); }); diff --git a/packages/ai/test/google-system-prompt.test.ts b/packages/ai/test/google-system-prompt.test.ts index 1ca718188..f5ef242be 100644 --- a/packages/ai/test/google-system-prompt.test.ts +++ b/packages/ai/test/google-system-prompt.test.ts @@ -26,6 +26,8 @@ async function captureGooglePayload( await streamGoogle(model, context, { apiKey: "test-key", + // Capture the generateContent request shape; gemini-3 ids auto-route to Interactions by default. + useInteractionsApi: false, onPayload: payload => { captured = payload as { config: { systemInstruction?: unknown }; contents: unknown[] }; }, diff --git a/packages/ai/test/issue-1270-repro.test.ts b/packages/ai/test/issue-1270-repro.test.ts index 5d3582254..2154ff79a 100644 --- a/packages/ai/test/issue-1270-repro.test.ts +++ b/packages/ai/test/issue-1270-repro.test.ts @@ -44,6 +44,8 @@ describe("issue #1270: Vertex AI global endpoint", () => { const stream = streamGoogleVertex(model, context, { project: "vertex-project", location: "global", + // This asserts the generateContent URL; gemini-3 ids auto-route to Interactions by default. + useInteractionsApi: false, fetch: async input => { const url = input instanceof Request ? input.url : input.toString(); urls.push(url); From bebdd22e64d669a2162df75c410b793d5ee8251a Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 06:37:49 +0200 Subject: [PATCH 36/88] fix(ai): healed leaked reasoning fences live for every provider - Added `leaked-thinking-stream.ts` whose `wrapLeakedThinkingStream`/`LeakedThinkingProjector` re-projects any provider stream, splitting leaked ```thinking`/`` fences out of the visible-text channel into structured thinking blocks live while preserving text/thinking/tool signatures. - Wrapped the shared `withProviderInFlightLimit` dispatch (both the unlimited and queued paths) and the standalone GitLab Duo path in `stream.ts` so healing covers every provider exit idempotently. - Reworked `google-gemini-cli.ts` visible-text emission onto `StreamMarkupHealing` with `feedVisibleText`/`flushVisibleText` and explicit block bookkeeping, lifting leaked Gemini fences ahead of native tool calls. - Added `leaked-thinking-stream` coverage and a gemini-cli healing case, and updated `stream-auth-retry`/`google-gemini-cli-alignment` expectations for the healed event sequence and dropped empty-text residue. - Recorded the changelog `### Fixed` entry (whose run also carries the adjacent Codex `all_turns` line). --- packages/ai/CHANGELOG.md | 4 +- .../ai/src/providers/google-gemini-cli.ts | 131 +++++--- packages/ai/src/stream.ts | 19 +- .../ai/src/utils/leaked-thinking-stream.ts | 260 ++++++++++++++++ .../google-function-calling-matching.test.ts | 145 +++++++++ .../test/google-gemini-cli-alignment.test.ts | 28 +- .../ai/test/leaked-thinking-stream.test.ts | 285 ++++++++++++++++++ packages/ai/test/stream-auth-retry.test.ts | 14 +- .../ai/test/stream-markup-healing.test.ts | 89 +++++- 9 files changed, 901 insertions(+), 74 deletions(-) create mode 100644 packages/ai/src/utils/leaked-thinking-stream.ts create mode 100644 packages/ai/test/google-function-calling-matching.test.ts create mode 100644 packages/ai/test/leaked-thinking-stream.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index ae67a9d21..b3e3e89e5 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -14,9 +14,9 @@ - Updated service tier logic to avoid global scopes in favor of per-provider configurations - Refactored priority request billing to better align with specific provider capabilities - Updated internal `coerceServiceTierByFamily` helper to facilitate migration from legacy settings +- Changed API-key resolution precedence so an explicit environment variable (e.g. `GEMINI_API_KEY`) overrides a stored/broker-migrated static API key; a deliberate OAuth login still takes precedence over the env var. ### Fixed -- Changed API-key resolution precedence so an explicit environment variable (e.g. `GEMINI_API_KEY`) overrides a stored/broker-migrated static API key; a deliberate OAuth login still takes precedence over the env var. - Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors @@ -24,6 +24,8 @@ - Ensured Gemini service tier is correctly passed through to the API - Corrected priority request accounting for supported providers - Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) +- Healed leaked reasoning fences (` ```thinking ` / ``) live for every provider via a central stream wrapper, splitting them into structured thinking blocks during streaming. +- Fixed Codex requests failing with `Unsupported value: 'all_turns' is not supported with this model`: the `reasoning.context: "all_turns"` default is now gated to gpt-5.4+ Codex models. Older ids (`gpt-5.1-codex`, `gpt-5.3-codex`, `gpt-5.3-codex-spark`) omit `context` so the server applies its `current_turn` default; an explicit `all_turns` override is also suppressed on those models, while `current_turn`/`auto` always pass through. ## [16.2.6] - 2026-06-29 diff --git a/packages/ai/src/providers/google-gemini-cli.ts b/packages/ai/src/providers/google-gemini-cli.ts index 9338eb972..adbe02a87 100644 --- a/packages/ai/src/providers/google-gemini-cli.ts +++ b/packages/ai/src/providers/google-gemini-cli.ts @@ -35,6 +35,7 @@ import { armPreResponseTimeout, getStreamFirstEventTimeoutMs } from "../utils/id // Refresh is the sole responsibility of AuthStorage (broker-aware, single-flighted); // the stream provider trusts the access token threaded through `options.apiKey`. import { normalizeSchemaForCCA } from "../utils/schema"; +import { StreamMarkupHealing, type StreamMarkupHealingEvent } from "../utils/stream-markup-healing"; import type { Content, FunctionCallingConfigMode, ThinkingConfig } from "./google-shared"; import { convertMessages, @@ -665,15 +666,43 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( let currentBlock: TextContent | ThinkingContent | null = null; const blocks = output.content; const blockIndex = () => blocks.length - 1; + const visibleTextHealing = new StreamMarkupHealing({ pattern: "thinking" }); let isBuffering = false; let textBuffer = ""; let bufferedTextSignature: string | undefined; - const emitVisibleText = (delta: string, thoughtSignature?: string) => { - if (!delta || !currentBlock || currentBlock.type !== "text") return; - currentBlock.text += delta; - currentBlock.textSignature = retainThoughtSignature(currentBlock.textSignature, thoughtSignature); + const endCurrentBlock = (): void => { + if (!currentBlock) return; + pushBlockEndEvent(currentBlock, blockIndex(), output, stream); + currentBlock = null; + }; + + const startTextBlock = (): TextContent => { + let block = currentBlock; + if (block?.type !== "text") { + endCurrentBlock(); + block = startTextOrThinkingBlock(false, output, stream, ensureStarted); + currentBlock = block; + } + return block; + }; + + const startThinkingBlock = (): ThinkingContent => { + let block = currentBlock; + if (block?.type !== "thinking") { + endCurrentBlock(); + block = startTextOrThinkingBlock(true, output, stream, ensureStarted); + currentBlock = block; + } + return block; + }; + + const emitVisibleText = (delta: string, thoughtSignature?: string): void => { + if (!delta) return; + const block = startTextBlock(); + block.text += delta; + block.textSignature = retainThoughtSignature(block.textSignature, thoughtSignature); stream.push({ type: "text_delta", contentIndex: blockIndex(), @@ -682,6 +711,48 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( }); }; + const emitVisibleThinking = (delta: string): void => { + if (!delta) return; + const block = startThinkingBlock(); + block.thinking += delta; + stream.push({ + type: "thinking_delta", + contentIndex: blockIndex(), + delta, + partial: output, + }); + }; + + const emitHealingEvent = (event: StreamMarkupHealingEvent, thoughtSignature?: string): void => { + if (event.type === "text") { + emitVisibleText(event.text, thoughtSignature); + } else if (event.type === "thinking") { + emitVisibleThinking(event.thinking); + } + }; + + const feedVisibleText = (delta: string, thoughtSignature?: string): void => { + for (const event of visibleTextHealing.feedEvents(delta)) { + emitHealingEvent(event, thoughtSignature); + } + }; + + const flushVisibleText = (thoughtSignature?: string): void => { + for (const event of visibleTextHealing.flushEvents()) { + emitHealingEvent(event, thoughtSignature); + } + }; + + const retainCurrentBlockThoughtSignature = (thoughtSignature: string): void => { + const block = currentBlock; + if (!block) return; + if (block.type === "thinking") { + block.thinkingSignature = retainThoughtSignature(block.thinkingSignature, thoughtSignature); + } else { + block.textSignature = retainThoughtSignature(block.textSignature, thoughtSignature); + } + }; + for await (const chunk of readSseJson( activeResponse.body!, options?.signal, @@ -710,20 +781,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( for (const part of candidate.content.parts) { if (part.text !== undefined && part.text !== "") { const isThinking = isThinkingPart(part); - if ( - !currentBlock || - (isThinking && currentBlock.type !== "thinking") || - (!isThinking && currentBlock.type !== "text") - ) { - if (currentBlock) { - pushBlockEndEvent(currentBlock, blockIndex(), output, stream); - } - currentBlock = startTextOrThinkingBlock(isThinking, output, stream, ensureStarted); - } - if (currentBlock.type === "thinking") { - currentBlock.thinking += part.text; - currentBlock.thinkingSignature = retainThoughtSignature( - currentBlock.thinkingSignature, + if (isThinking) { + flushVisibleText(); + const block = startThinkingBlock(); + block.thinking += part.text; + block.thinkingSignature = retainThoughtSignature( + block.thinkingSignature, part.thoughtSignature, ); stream.push({ @@ -744,7 +807,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( textBuffer = part.text; bufferedTextSignature = part.thoughtSignature; } else { - emitVisibleText(part.text, part.thoughtSignature); + feedVisibleText(part.text, part.thoughtSignature); } if (isBuffering) { @@ -757,32 +820,19 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( isBuffering = false; textBuffer = ""; bufferedTextSignature = undefined; - emitVisibleText(buffered.visibleText, visibleSignature); + feedVisibleText(buffered.visibleText, visibleSignature); } } } - } else if (part.text === "" && part.thoughtSignature && currentBlock && !part.functionCall) { - if (currentBlock.type === "thinking") { - currentBlock.thinkingSignature = retainThoughtSignature( - currentBlock.thinkingSignature, - part.thoughtSignature, - ); - } else { - currentBlock.textSignature = retainThoughtSignature( - currentBlock.textSignature, - part.thoughtSignature, - ); - } + } else if (part.text === "" && part.thoughtSignature && !part.functionCall) { + retainCurrentBlockThoughtSignature(part.thoughtSignature); } if (part.functionCall) { - if (currentBlock) { - pushBlockEndEvent(currentBlock, blockIndex(), output, stream); - currentBlock = null; - } + flushVisibleText(); + endCurrentBlock(); isBuffering = false; textBuffer = ""; - const providedId = part.functionCall.id; const needsNewId = !providedId || output.content.some(b => b.type === "toolCall" && b.id === providedId); @@ -848,16 +898,15 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( sawLeak = true; } if (buffered.kind !== "incomplete") { - emitVisibleText(buffered.visibleText, bufferedTextSignature); + feedVisibleText(buffered.visibleText, bufferedTextSignature); } bufferedTextSignature = undefined; isBuffering = false; textBuffer = ""; } - if (currentBlock) { - pushBlockEndEvent(currentBlock, blockIndex(), output, stream); - } + flushVisibleText(bufferedTextSignature); + endCurrentBlock(); return hasMeaningfulGoogleContent(output) || sawLeak; }; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 102a7c717..62d572ba0 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -71,6 +71,7 @@ import type { ToolChoice, } from "./types"; import { AssistantMessageEventStream } from "./utils/event-stream"; +import { wrapLeakedThinkingStream } from "./utils/leaked-thinking-stream"; import { wrapFetchForProxy } from "./utils/proxy"; import { withRequestDebugFetch } from "./utils/request-debug"; import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop"; @@ -499,8 +500,11 @@ function withProviderInFlightLimit AssistantMessageEventStream, ): AssistantMessageEventStream { + // Leaked-thinking healing folds in here — the one shared provider-dispatch + // chokepoint — so the loop guard (which wraps this) sees healed events and all + // six provider exits are covered by one wrap. Healing is idempotent. const limit = resolveProviderInFlightLimit(model.provider, options); - if (limit === undefined) return dispatch(); + if (limit === undefined) return wrapLeakedThinkingStream(dispatch()); const outer = new AssistantMessageEventStream(); void (async () => { @@ -520,7 +524,7 @@ function withProviderInFlightLimit( // GitLab Duo Workflow - IDE workflow protocol + WebSocket action bridge if (model.api === "gitlab-duo-agent") { - return streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, { - ...requestOptions, - apiKey, - }); + // Does not route through withProviderInFlightLimit, so heal explicitly. + return wrapLeakedThinkingStream( + streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, { + ...requestOptions, + apiKey, + }), + ); } // Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API diff --git a/packages/ai/src/utils/leaked-thinking-stream.ts b/packages/ai/src/utils/leaked-thinking-stream.ts new file mode 100644 index 000000000..9596a9b73 --- /dev/null +++ b/packages/ai/src/utils/leaked-thinking-stream.ts @@ -0,0 +1,260 @@ +/** + * Central live healing for leaked reasoning markup in the visible text channel. + * + * Some providers emit their canonical reasoning idioms (` ```thinking `, + * ``, Gemma/Harmony channels, …) into the *visible* text stream instead + * of a structured thinking part. {@link wrapLeakedThinkingStream} re-projects any + * provider stream into a fresh {@link AssistantMessageEventStream}, splitting the + * leaked fences out into proper `thinking` blocks *live* as deltas arrive — so + * every provider gets the same healing, not just the three with provider-local + * {@link StreamMarkupHealing} loops. + * + * The healing is idempotent: a second pass over already-clean text finds no + * fences, so wrapping a provider that already heals (or wrapping twice) is a + * harmless pass-through. Signatures are load-bearing for Google/Gemini/Vertex + * thought round-tripping, so text sub-blocks carry the source `textSignature`, + * forwarded thinking blocks their `thinkingSignature`, and forwarded tool calls + * their `thoughtSignature`. + * + * Modeled on {@link wrapInbandToolStream} / `InbandStreamProjector` in + * `../dialect/owned-stream.ts`, minus all in-band tool-call grammar: tool-call + * events are forwarded verbatim. + */ + +import type { AssistantMessage, TextContent, ThinkingContent, ToolCall } from "../types"; +import { AssistantMessageEventStream } from "./event-stream"; +import { StreamMarkupHealing, type StreamMarkupHealingEvent } from "./stream-markup-healing"; + +/** + * Wrap a provider stream so leaked reasoning fences are healed into thinking + * blocks live, for every provider. Returns a new stream that re-projects the + * inner one; the inner stream is fully consumed. + */ +export function wrapLeakedThinkingStream(inner: AssistantMessageEventStream): AssistantMessageEventStream { + const out = new AssistantMessageEventStream(); + void (async () => { + try { + let projector: LeakedThinkingProjector | undefined; + for await (const event of inner) { + switch (event.type) { + case "start": + projector = new LeakedThinkingProjector(out, event.partial); + break; + case "text_delta": { + projector ??= new LeakedThinkingProjector(out, event.partial); + const block = event.partial.content[event.contentIndex]; + projector.text(event.delta, block?.type === "text" ? block.textSignature : undefined); + break; + } + case "thinking_delta": { + projector ??= new LeakedThinkingProjector(out, event.partial); + const block = event.partial.content[event.contentIndex]; + projector.thinking(event.delta, block?.type === "thinking" ? block.thinkingSignature : undefined); + break; + } + case "toolcall_start": { + projector ??= new LeakedThinkingProjector(out, event.partial); + const block = event.partial.content[event.contentIndex]; + projector.toolStart(event.contentIndex, block?.type === "toolCall" ? block.name : ""); + break; + } + case "toolcall_delta": + projector?.toolDelta(event.contentIndex, event.delta); + break; + case "toolcall_end": + projector?.toolEnd(event.contentIndex, event.toolCall); + break; + case "done": { + projector ??= new LeakedThinkingProjector(out, event.message); + const content = projector.finish(event.message); + out.push({ type: "done", reason: event.reason, message: { ...event.message, content } }); + return; + } + case "error": { + projector ??= new LeakedThinkingProjector(out, event.error); + const content = projector.finish(event.error); + out.push({ type: "error", reason: event.reason, error: { ...event.error, content } }); + return; + } + // text_start/text_end/thinking_start/thinking_end are ignored: the + // projector owns block boundaries (matches wrapInbandToolStream). + } + } + // Inner ended via end(result) without a terminal event. + if (!out.done) { + const result = await inner.result(); + projector ??= new LeakedThinkingProjector(out, result); + const content = projector.finish(result); + out.end({ ...result, content }); + } + } catch (err) { + if (!out.done) out.fail(err); + } + })(); + return out; +} + +type OpenBlock = { index: number } | undefined; + +/** + * Re-projects an inner stream's events into `out`, healing leaked reasoning out + * of the visible text channel while forwarding native thinking and tool calls. + */ +class LeakedThinkingProjector { + readonly #out: AssistantMessageEventStream; + readonly #healer = new StreamMarkupHealing({ pattern: "thinking" }); + #partial: AssistantMessage; + #text: OpenBlock; + #thinking: OpenBlock; + /** Total visible text length fed to the healer, to replay any un-streamed tail in {@link finish}. */ + #fedLen = 0; + /** Latest non-undefined text signature seen, stamped onto held-back text flushed later. */ + #lastTextSignature: string | undefined; + /** Forwarded native tool calls, keyed by the inner stream's `contentIndex`. */ + #toolBlocks = new Map(); + + constructor(out: AssistantMessageEventStream, seed: AssistantMessage) { + this.#out = out; + this.#partial = { ...seed, content: [] }; + this.#out.push({ type: "start", partial: this.#partial }); + } + + /** Feed a visible-text delta through the healer, splitting leaked fences live. */ + text(delta: string, signature: string | undefined): void { + this.#fedLen += delta.length; + if (signature !== undefined) this.#lastTextSignature = signature; + this.#apply(this.#healer.feedEvents(delta), this.#lastTextSignature); + } + + /** Forward a native thinking delta, preserving its signature. */ + thinking(delta: string, signature: string | undefined): void { + const index = this.#openThinking(); + const block = this.#partial.content[index] as ThinkingContent; + block.thinking += delta; + if (signature !== undefined) block.thinkingSignature = signature; + this.#out.push({ type: "thinking_delta", contentIndex: index, delta, partial: this.#partial }); + } + + /** Forward a native tool call's start, releasing any held-back text first. */ + toolStart(srcIndex: number, name: string): void { + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + const block: ToolCall = { type: "toolCall", id: "", name, arguments: {} }; + this.#partial.content.push(block); + const index = this.#partial.content.length - 1; + this.#toolBlocks.set(srcIndex, { index }); + this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial }); + } + + toolDelta(srcIndex: number, delta: string): void { + const entry = this.#toolBlocks.get(srcIndex); + if (!entry) return; + this.#out.push({ type: "toolcall_delta", contentIndex: entry.index, delta, partial: this.#partial }); + } + + toolEnd(srcIndex: number, toolCall: ToolCall): void { + const entry = this.#toolBlocks.get(srcIndex); + if (entry) { + const block = this.#partial.content[entry.index] as ToolCall; + Object.assign(block, toolCall); + this.#out.push({ type: "toolcall_end", contentIndex: entry.index, toolCall: block, partial: this.#partial }); + this.#toolBlocks.delete(srcIndex); + return; + } + // `end` without a matching `start` — release held text, then forward whole. + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + const block: ToolCall = { ...toolCall }; + this.#partial.content.push(block); + const index = this.#partial.content.length - 1; + this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial }); + this.#out.push({ type: "toolcall_end", contentIndex: index, toolCall: block, partial: this.#partial }); + } + + /** + * Finalize: replay any un-streamed visible-text tail from `message.content`, + * flush held-back fragments, close open blocks, and return the healed content. + */ + finish(message: AssistantMessage): AssistantMessage["content"] { + let fullText = ""; + let tailSignature: string | undefined; + for (const block of message.content) { + if (block.type === "text") { + fullText += block.text; + tailSignature = block.textSignature; + } + } + if (tailSignature !== undefined) this.#lastTextSignature = tailSignature; + if (fullText.length > this.#fedLen) { + this.#apply(this.#healer.feedEvents(fullText.slice(this.#fedLen)), this.#lastTextSignature); + } + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + return this.#partial.content; + } + + #apply(events: readonly StreamMarkupHealingEvent[], signature?: string): void { + for (const event of events) { + if (event.type === "text") this.#emitText(event.text, signature); + else if (event.type === "thinking") this.#emitHealedThinking(event.thinking); + } + } + + #emitText(text: string, signature: string | undefined): void { + if (text.length === 0) return; + this.#closeThinking(); + if (!this.#text) { + const block: TextContent = + signature === undefined ? { type: "text", text: "" } : { type: "text", text: "", textSignature: signature }; + this.#partial.content.push(block); + this.#text = { index: this.#partial.content.length - 1 }; + this.#out.push({ type: "text_start", contentIndex: this.#text.index, partial: this.#partial }); + } else if (signature !== undefined) { + (this.#partial.content[this.#text.index] as TextContent).textSignature = signature; + } + const block = this.#partial.content[this.#text.index] as TextContent; + block.text += text; + this.#out.push({ type: "text_delta", contentIndex: this.#text.index, delta: text, partial: this.#partial }); + } + + /** Healed (leaked) thinking carries no signature, matching the source fence. */ + #emitHealedThinking(text: string): void { + if (text.length === 0) return; + const index = this.#openThinking(); + const block = this.#partial.content[index] as ThinkingContent; + block.thinking += text; + this.#out.push({ type: "thinking_delta", contentIndex: index, delta: text, partial: this.#partial }); + } + + #openThinking(): number { + this.#closeText(); + if (!this.#thinking) { + this.#partial.content.push({ type: "thinking", thinking: "" }); + this.#thinking = { index: this.#partial.content.length - 1 }; + this.#out.push({ type: "thinking_start", contentIndex: this.#thinking.index, partial: this.#partial }); + } + return this.#thinking.index; + } + + #closeText(): void { + if (!this.#text) return; + const block = this.#partial.content[this.#text.index] as TextContent; + this.#out.push({ type: "text_end", contentIndex: this.#text.index, content: block.text, partial: this.#partial }); + this.#text = undefined; + } + + #closeThinking(): void { + if (!this.#thinking) return; + const block = this.#partial.content[this.#thinking.index] as ThinkingContent; + this.#out.push({ + type: "thinking_end", + contentIndex: this.#thinking.index, + content: block.thinking, + partial: this.#partial, + }); + this.#thinking = undefined; + } +} diff --git a/packages/ai/test/google-function-calling-matching.test.ts b/packages/ai/test/google-function-calling-matching.test.ts new file mode 100644 index 000000000..9f610e19e --- /dev/null +++ b/packages/ai/test/google-function-calling-matching.test.ts @@ -0,0 +1,145 @@ +import { describe, expect, it } from "bun:test"; +import { convertMessages } from "@oh-my-pi/pi-ai/providers/google-shared"; +import type { Context, Model, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const ZERO_USAGE: Usage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function createGoogleModel( + id: string, + api: "google-generative-ai" | "google-vertex" = "google-generative-ai", +): Model { + return buildModel({ + id, + name: id, + api, + provider: api === "google-vertex" ? "google-vertex" : "google", + baseUrl: "https://example.com", + reasoning: false, + input: ["text", "image"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 200000, + maxTokens: 8192, + }); +} + +function contextWithToolResult(toolName = "stale_tool_name"): Context { + return { + messages: [ + { + role: "user", + content: "Call the tool", + timestamp: 1000, + }, + { + role: "assistant", + provider: "google-generative-ai", + api: "google-generative-ai", + model: "gemini-2.5-flash", + content: [ + { + type: "toolCall", + id: "call_12345_abc", + name: "actual_tool_name", + arguments: { query: "pi" }, + }, + ], + usage: ZERO_USAGE, + stopReason: "toolUse", + timestamp: 2000, + }, + { + role: "toolResult", + toolCallId: "call_12345_abc", + toolName, + isError: false, + content: [ + { + type: "text", + text: "Tool result text", + }, + ], + timestamp: 3000, + }, + ], + }; +} + +function functionCallAndResponse(model: Model<"google-generative-ai" | "google-vertex">, context: Context) { + const contents = convertMessages(model, context); + const functionCall = contents.find(c => c.role === "model")?.parts?.find(part => part.functionCall)?.functionCall; + const functionResponse = contents + .find(c => c.role === "user" && c.parts?.some(part => part.functionResponse)) + ?.parts?.find(part => part.functionResponse)?.functionResponse; + + return { functionCall, functionResponse }; +} + +describe("Google GenerateContent function response matching", () => { + it("uses emitted functionCall IDs and names for direct Gemini 3 functionResponse parts", () => { + const model = createGoogleModel("gemini-3.5-flash"); + const { functionCall, functionResponse } = functionCallAndResponse(model, contextWithToolResult()); + + expect(functionCall?.id).toBe("call_12345_abc"); + expect(functionResponse?.id).toBe("call_12345_abc"); + expect(functionCall?.name).toBe("actual_tool_name"); + expect(functionResponse?.name).toBe("actual_tool_name"); + expect(functionResponse?.name).toBe(functionCall?.name); + }); + + it("omits unsupported Part IDs for Vertex Gemini 3.5 GenerateContent", () => { + const model = createGoogleModel("gemini-3.5-flash", "google-vertex"); + const { functionCall, functionResponse } = functionCallAndResponse(model, contextWithToolResult()); + + expect(functionCall?.id).toBeUndefined(); + expect(functionResponse?.id).toBeUndefined(); + expect(functionResponse?.name).toBe(functionCall?.name); + }); + + it("keeps multimodal tool output inside Gemini 3 functionResponse parts", () => { + const context = contextWithToolResult("actual_tool_name"); + const toolResult = context.messages[2]; + if (toolResult.role !== "toolResult") throw new Error("expected tool result fixture"); + toolResult.content.push({ + type: "image", + mimeType: "image/png", + data: "base64-image-data", + }); + + const model = createGoogleModel("gemini-3.5-flash"); + const { functionResponse } = functionCallAndResponse(model, context); + + expect(functionResponse?.parts).toEqual([ + { + inlineData: { + mimeType: "image/png", + data: "base64-image-data", + }, + }, + ]); + }); + + it("keeps Claude call IDs on non-Vertex Google-compatible endpoints", () => { + const model = createGoogleModel("claude-sonnet-4-5"); + const { functionCall, functionResponse } = functionCallAndResponse( + model, + contextWithToolResult("actual_tool_name"), + ); + + expect(functionCall?.id).toBe("call_12345_abc"); + expect(functionResponse?.id).toBe("call_12345_abc"); + expect(functionResponse?.name).toBe(functionCall?.name); + }); +}); diff --git a/packages/ai/test/google-gemini-cli-alignment.test.ts b/packages/ai/test/google-gemini-cli-alignment.test.ts index b7de79719..3e8c2f463 100644 --- a/packages/ai/test/google-gemini-cli-alignment.test.ts +++ b/packages/ai/test/google-gemini-cli-alignment.test.ts @@ -596,11 +596,9 @@ describe("Google Gemini CLI alignment", () => { } const result = await stream.result(); - expect(result.content).toHaveLength(1); - expect(result.content[0]).toEqual({ - type: "text", - text: "", - }); + // A fully-discarded planning leak leaves no residual content — no empty + // text block survives (the central healing wrapper strips empties too). + expect(result.content).toHaveLength(0); expect(result.stopReason).toBe("stop"); const textDeltaEvents = events.filter(e => e.type === "text_delta"); @@ -838,15 +836,11 @@ describe("Google Gemini CLI alignment", () => { } const result = await stream.result(); - expect(result.content).toHaveLength(2); - expect(result.content[0]).toEqual({ - type: "text", - text: "", - }); - expect(result.content[1].type).toBe("toolCall"); - if (result.content[1].type === "toolCall") { - expect(result.content[1].name).toBe("read"); - expect(result.content[1].arguments).toEqual({ path: "src/main.ts" }); + expect(result.content).toHaveLength(1); + expect(result.content[0].type).toBe("toolCall"); + if (result.content[0].type === "toolCall") { + expect(result.content[0].name).toBe("read"); + expect(result.content[0].arguments).toEqual({ path: "src/main.ts" }); } expect(events.filter(e => e.type === "toolcall_start")).toHaveLength(1); @@ -917,11 +911,7 @@ describe("Google Gemini CLI alignment", () => { events.push(event); } const result = await stream.result(); - expect(result.content).toHaveLength(1); - expect(result.content[0]).toEqual({ - type: "text", - text: "", - }); + expect(result.content).toHaveLength(0); }); }); }); diff --git a/packages/ai/test/leaked-thinking-stream.test.ts b/packages/ai/test/leaked-thinking-stream.test.ts new file mode 100644 index 000000000..d2b57166c --- /dev/null +++ b/packages/ai/test/leaked-thinking-stream.test.ts @@ -0,0 +1,285 @@ +import { describe, expect, it } from "bun:test"; +import { stream } from "@oh-my-pi/pi-ai/stream"; +import type { + AssistantMessage, + AssistantMessageEvent, + Context, + FetchImpl, + Model, + TextContent, + ThinkingContent, + ToolCall, +} from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { wrapLeakedThinkingStream } from "@oh-my-pi/pi-ai/utils/leaked-thinking-stream"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +/** Minimal assistant message; `content`/`stopReason` overridden per event. */ +function msg(overrides: Partial = {}): AssistantMessage { + return { + role: "assistant", + content: [], + api: "mock", + provider: "mock", + model: "mock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 0, + ...overrides, + }; +} + +/** + * Drive the wrapper: push inner events synchronously, then drain the healed + * output. Returns every emitted event plus the resolved final message. + */ +async function runWrapper( + feed: (inner: AssistantMessageEventStream) => void, +): Promise<{ events: AssistantMessageEvent[]; result: AssistantMessage }> { + const inner = new AssistantMessageEventStream(); + const out = wrapLeakedThinkingStream(inner); + feed(inner); + const events: AssistantMessageEvent[] = []; + for await (const event of out) events.push(event); + const result = await out.result(); + return { events, result }; +} + +function texts(message: AssistantMessage): string[] { + return message.content.filter((b): b is TextContent => b.type === "text").map(b => b.text); +} + +function thinks(message: AssistantMessage): ThinkingContent[] { + return message.content.filter((b): b is ThinkingContent => b.type === "thinking"); +} + +describe("wrapLeakedThinkingStream", () => { + it("splits a leaked fence into structured blocks live during streaming", async () => { + const leaked = "Visible before.```thinking\nplan\n```Visible after."; + const { events, result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ type: "text_start", contentIndex: 0, partial: msg({ content: [{ type: "text", text: "" }] }) }); + inner.push({ + type: "text_delta", + contentIndex: 0, + delta: leaked, + partial: msg({ content: [{ type: "text", text: leaked }] }), + }); + inner.push({ + type: "text_end", + contentIndex: 0, + content: leaked, + partial: msg({ content: [{ type: "text", text: leaked }] }), + }); + inner.push({ type: "done", reason: "stop", message: msg({ content: [{ type: "text", text: leaked }] }) }); + }); + + expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]); + expect(texts(result)).toEqual(["Visible before.", "Visible after."]); + expect(thinks(result).map(b => b.thinking)).toEqual(["plan\n"]); + // The split happened live, not only in the terminal message. + expect(events.some(e => e.type === "thinking_delta")).toBe(true); + }); + + it("preserves text, thinking, and tool-call signatures across the split", async () => { + const leaked = "before ```thinking\nhmm\n``` after"; + const call: ToolCall = { + type: "toolCall", + id: "call_1", + name: "read", + arguments: { path: "x" }, + thoughtSignature: "tsig", + }; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "text_start", + contentIndex: 0, + partial: msg({ content: [{ type: "text", text: "", textSignature: "sig" }] }), + }); + inner.push({ + type: "text_delta", + contentIndex: 0, + delta: leaked, + partial: msg({ content: [{ type: "text", text: leaked, textSignature: "sig" }] }), + }); + const withCall = msg({ + content: [{ type: "text", text: leaked, textSignature: "sig" }, call], + stopReason: "toolUse", + }); + inner.push({ type: "toolcall_start", contentIndex: 1, partial: withCall }); + inner.push({ type: "toolcall_end", contentIndex: 1, toolCall: call, partial: withCall }); + inner.push({ type: "done", reason: "toolUse", message: withCall }); + }); + + const textBlocks = result.content.filter((b): b is TextContent => b.type === "text"); + expect(textBlocks.map(b => b.text)).toEqual(["before ", " after"]); + expect(textBlocks.map(b => b.textSignature)).toEqual(["sig", "sig"]); + // Healed (leaked) thinking carries no signature. + expect(thinks(result).every(b => b.thinkingSignature === undefined)).toBe(true); + const calls = result.content.filter((b): b is ToolCall => b.type === "toolCall"); + expect(calls[0]?.thoughtSignature).toBe("tsig"); + }); + + it("heals a fence that only appears in the terminal message (no prior text deltas)", async () => { + const leaked = "Intro.```thinking\nquiet\n```Outro."; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "done", + reason: "stop", + message: msg({ content: [{ type: "text", text: leaked, textSignature: "sig" }] }), + }); + }); + + expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]); + expect(texts(result)).toEqual(["Intro.", "Outro."]); + // Tail-replayed text still carries the source signature. + expect(result.content.filter((b): b is TextContent => b.type === "text").map(b => b.textSignature)).toEqual([ + "sig", + "sig", + ]); + }); + + it("passes clean text through unchanged and forwards native thinking", async () => { + const clean = "Just a normal answer."; + const cleanRun = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ type: "text_start", contentIndex: 0, partial: msg({ content: [{ type: "text", text: "" }] }) }); + inner.push({ + type: "text_delta", + contentIndex: 0, + delta: clean, + partial: msg({ content: [{ type: "text", text: clean }] }), + }); + inner.push({ type: "done", reason: "stop", message: msg({ content: [{ type: "text", text: clean }] }) }); + }); + expect(cleanRun.result.content.map(b => b.type)).toEqual(["text"]); + expect(texts(cleanRun.result)).toEqual([clean]); + + const nativeThinking = msg({ + content: [ + { type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }, + { type: "text", text: "answer" }, + ], + }); + const nativeRun = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "thinking_start", + contentIndex: 0, + partial: msg({ content: [{ type: "thinking", thinking: "" }] }), + }); + inner.push({ + type: "thinking_delta", + contentIndex: 0, + delta: "native reasoning", + partial: msg({ content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }] }), + }); + inner.push({ + type: "thinking_end", + contentIndex: 0, + content: "native reasoning", + partial: msg({ content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }] }), + }); + inner.push({ type: "text_start", contentIndex: 1, partial: nativeThinking }); + inner.push({ type: "text_delta", contentIndex: 1, delta: "answer", partial: nativeThinking }); + inner.push({ type: "done", reason: "stop", message: nativeThinking }); + }); + expect(nativeRun.result.content.map(b => b.type)).toEqual(["thinking", "text"]); + expect(thinks(nativeRun.result)[0]?.thinking).toBe("native reasoning"); + expect(thinks(nativeRun.result)[0]?.thinkingSignature).toBe("tk"); + expect(texts(nativeRun.result)).toEqual(["answer"]); + }); + + it("heals a terminal error message and keeps its error stop reason", async () => { + const leaked = "Partial.```thinking\noops\n```Recovered."; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "error", + reason: "error", + error: msg({ content: [{ type: "text", text: leaked }], stopReason: "error" }), + }); + }); + + expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]); + expect(texts(result)).toEqual(["Partial.", "Recovered."]); + expect(result.stopReason).toBe("error"); + }); +}); + +describe("leaked thinking healing through stream()", () => { + function sseFrame(event: string, data: unknown): string { + return `event: ${event}\ndata: ${JSON.stringify(data)}\n\n`; + } + + function anthropicLeakFetch(text: string): FetchImpl { + const body = [ + sseFrame("message_start", { + type: "message_start", + message: { id: "msg_leak", usage: { input_tokens: 5, output_tokens: 0 } }, + }), + sseFrame("content_block_start", { + type: "content_block_start", + index: 0, + content_block: { type: "text", text: "" }, + }), + sseFrame("content_block_delta", { + type: "content_block_delta", + index: 0, + delta: { type: "text_delta", text }, + }), + sseFrame("content_block_stop", { type: "content_block_stop", index: 0 }), + sseFrame("message_delta", { + type: "message_delta", + delta: { stop_reason: "end_turn" }, + usage: { input_tokens: 5, output_tokens: 4 }, + }), + sseFrame("message_stop", { type: "message_stop" }), + ].join(""); + const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise => + new Response(body, { + status: 200, + headers: { "content-type": "text/event-stream", "request-id": "req_mock" }, + }); + return Object.assign(fn, { preconnect: fetch.preconnect }); + } + + it("splits a leaked fence from a provider with no own healer", async () => { + // Anthropic has no provider-local visible-text healer, so a split here + // proves the central wrapper is composed into stream(). + const model: Model<"anthropic-messages"> = buildModel({ + id: "claude-sonnet-4-5", + name: "Claude Sonnet 4.5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }); + const leaked = "```thinking\nDeliberate.\n```\nFinal answer."; + const context: Context = { messages: [{ role: "user", content: "hi", timestamp: Date.now() }] }; + const result = await stream(model, context, { + apiKey: "test", + fetch: anthropicLeakFetch(leaked), + }).result(); + + expect(result.content.map(b => b.type)).toEqual(["thinking", "text"]); + const thinking = thinks(result) + .map(b => b.thinking) + .join(""); + expect(thinking).toContain("Deliberate."); + expect(texts(result).join("").trim()).toBe("Final answer."); + }); +}); diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index 307b37412..484d80f01 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -153,9 +153,9 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "new-key"]); - // The buffered `start` of the failed attempt must not leak — the user - // sees exactly one clean start/done pair. - expect(eventTypes).toEqual(["start", "done"]); + // The failed attempt's buffered start must not leak — the user sees a + // single start from the successful attempt, then its healed content. + expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); }); it("retries on a 401 carried only via errorStatus", async () => { @@ -206,6 +206,12 @@ describe("streamSimple resolver auth retry", () => { queueMicrotask(() => { stream.push({ type: "start", partial: assistant() }); stream.push({ type: "text_start", contentIndex: 0, partial: assistant([""]) }); + stream.push({ + type: "text_delta", + contentIndex: 0, + delta: "partial", + partial: assistant(["partial"]), + }); stream.fail(failure); }); return stream; @@ -398,7 +404,7 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["credential-A", "credential-B"]); - expect(eventTypes).toEqual(["start", "done"]); + expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([false, true]); } }); diff --git a/packages/ai/test/stream-markup-healing.test.ts b/packages/ai/test/stream-markup-healing.test.ts index 27e72d26a..23f525eeb 100644 --- a/packages/ai/test/stream-markup-healing.test.ts +++ b/packages/ai/test/stream-markup-healing.test.ts @@ -5,9 +5,10 @@ import { type InbandScanEvent, ThinkingInbandScanner, } from "@oh-my-pi/pi-ai/dialect"; +import { streamGoogleGeminiCli } from "@oh-my-pi/pi-ai/providers/google-gemini-cli"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { stream } from "@oh-my-pi/pi-ai/stream"; -import type { Context, FetchImpl, Model, ThinkingContent, Tool, ToolCall } from "@oh-my-pi/pi-ai/types"; +import type { Context, FetchImpl, Model, TextContent, ThinkingContent, Tool, ToolCall } from "@oh-my-pi/pi-ai/types"; import { getStreamMarkupHealingPattern, StreamMarkupHealing } from "@oh-my-pi/pi-ai/utils/stream-markup-healing"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; @@ -38,7 +39,7 @@ interface SseChunk { }>; } -function sseResponse(events: ReadonlyArray): Response { +function sseResponse(events: ReadonlyArray): Response { const payload = `${events .map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`) .join("\n\n")}\n\n`; @@ -48,7 +49,7 @@ function sseResponse(events: ReadonlyArray): Response { }); } -function mockFetch(events: ReadonlyArray): FetchImpl { +function mockFetch(events: ReadonlyArray): FetchImpl { const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise => sseResponse(events); return Object.assign(fn, { preconnect: fetch.preconnect }); } @@ -124,6 +125,21 @@ const deepseekCloudModel: Model<"ollama-chat"> = buildModel({ maxTokens: 8_192, }); +function geminiCliModel(): Model<"google-gemini-cli"> { + return buildModel({ + id: "gemini-3.5-flash", + name: "Gemini 3.5 Flash", + api: "google-gemini-cli", + provider: "google-antigravity", + baseUrl: "https://antigravity.test", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8_192, + }); +} + function ndjsonResponse(lines: ReadonlyArray): Response { const body = `${lines.map(line => JSON.stringify(line)).join("\n")}\n`; const encoder = new TextEncoder(); @@ -230,6 +246,73 @@ describe("openai-completions leaked thinking healing", () => { }); }); +describe("google-gemini-cli leaked thinking healing", () => { + it("lifts a leaked Gemini thinking fence before a native tool call", async () => { + const model = geminiCliModel(); + const fetchMock = mockFetch([ + { + response: { + candidates: [ + { + content: { + role: "model", + parts: [ + { + text: "```thinking\nCheck the provider path.\n```\nI will inspect the file.", + thoughtSignature: "visible-text-signature", + }, + { + functionCall: { + name: "read", + args: { path: "packages/ai/src/providers/google-gemini-cli.ts" }, + id: "call_read_1", + }, + thoughtSignature: "function-call-signature", + }, + ], + }, + finishReason: "STOP", + }, + ], + usageMetadata: { + promptTokenCount: 10, + candidatesTokenCount: 5, + thoughtsTokenCount: 3, + totalTokenCount: 18, + }, + }, + }, + ]); + + const result = await streamGoogleGeminiCli( + model, + { ...baseContext(), tools: [readTool] }, + { + apiKey: JSON.stringify({ token: "test-token", projectId: "test-project" }), + fetch: fetchMock, + }, + ).result(); + + expect(result.content.map(block => block.type)).toEqual(["thinking", "text", "toolCall"]); + const thinking = result.content + .filter((block): block is ThinkingContent => block.type === "thinking") + .map(block => block.thinking) + .join(""); + const textBlocks = result.content.filter((block): block is TextContent => block.type === "text"); + const text = textBlocks.map(block => block.text).join(""); + const calls = result.content.filter((block): block is ToolCall => block.type === "toolCall"); + + expect(thinking).toBe("Check the provider path.\n"); + expect(text).toBe("\nI will inspect the file."); + expect(text).not.toContain("```thinking"); + expect(calls).toHaveLength(1); + expect(textBlocks[0]?.textSignature).toBe("visible-text-signature"); + expect(calls[0]?.id).toBe("call_read_1"); + expect(calls[0]?.thoughtSignature).toBe("function-call-signature"); + expect(result.stopReason).toBe("toolUse"); + }); +}); + describe("StreamMarkupHealing DSML envelope pattern", () => { it("parses the reporter's verbatim leak into a structured tool call", () => { const healing = new StreamMarkupHealing({ pattern: "dsml" }); From 093660f8b7cd4d79fd159aace9702486fcbb1a71 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 06:42:46 +0200 Subject: [PATCH 37/88] fix(ai): restricted reasoning context all_turns option to openai models v5.4 and newer - Guarded the `all_turns` reasoning context value to OpenAI models version 5.4 or greater. - Suppressed `reasoning.context` defaults and explicit overrides when `all_turns` is requested on unsupported models to prevent server rejection. - Introduced `supportsAllTurnsReasoningContext` helper in `@oh-my-pi/pi-catalog/identity` using semver classification. - Updated request-transformer and response-options logic to conditionalize the request payload shaping. - Expanded test coverage to verify correct fallback behavior and explicit overrides across gpt-5.x versions. --- .../src/providers/openai-codex-responses.ts | 2 +- .../openai-codex/request-transformer.ts | 21 +++++++-- .../test/openai-codex-responses-lite.test.ts | 47 ++++++++++++++++--- packages/catalog/src/identity/family.ts | 17 +++++++ 4 files changed, 75 insertions(+), 12 deletions(-) diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 2a15b7bc9..446727a9f 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -104,7 +104,7 @@ import { transformMessages } from "./transform-messages"; export interface OpenAICodexResponsesOptions extends StreamOptions { reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; reasoningSummary?: "auto" | "concise" | "detailed" | null; - /** `reasoning.context` replay scope; defaults to `all_turns` for every Codex request when unset. */ + /** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index b627fcbfa..77882a60e 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,4 +1,5 @@ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { supportsAllTurnsReasoningContext } from "@oh-my-pi/pi-catalog/identity"; import { requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking"; import type { Api, Model } from "../../types"; @@ -14,7 +15,7 @@ export interface ReasoningConfig { export interface CodexRequestOptions { reasoningEffort?: ReasoningConfig["effort"]; reasoningSummary?: ReasoningConfig["summary"] | null; - /** Explicit `reasoning.context` override; defaults to `all_turns` for every Codex request when unset. */ + /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; @@ -254,9 +255,21 @@ export async function transformRequestBody( ...body.reasoning, ...reasoningConfig, }; - // Default reasoning replay to `all_turns` for every Codex request, - // mirroring codex-rs; an explicit `reasoningContext` overrides it. - body.reasoning.context = options.reasoningContext ?? "all_turns"; + // Default reasoning replay to `all_turns`, mirroring codex-rs; an + // explicit `reasoningContext` overrides the default. The `all_turns` + // value is only accepted from gpt-5.4 onward — earlier Codex ids + // (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with + // "Unsupported value: 'all_turns' is not supported with this model". + // For those, drop `context` so the server applies its `current_turn` + // default. The version gate is authoritative: even an explicit + // `all_turns` override is suppressed on unsupported models, while + // `current_turn`/`auto` (universally supported) always pass through. + const context = options.reasoningContext ?? "all_turns"; + if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) { + delete body.reasoning.context; + } else { + body.reasoning.context = context; + } } else { delete body.reasoning; } diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 575c3cdd3..c6aee594f 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -83,21 +83,21 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe } describe("openai-codex reasoning.context", () => { - it("forwards an explicit reasoning.context and defaults to all_turns", async () => { - const model = createCodexModel("gpt-5.1-codex"); + it("defaults to all_turns on gpt-5.4+ models and forwards explicit overrides", async () => { + const model = createCodexModel("gpt-5.4"); + + const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); + expect(defaulted.reasoning?.context).toBe("all_turns"); const explicit = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", reasoningContext: "current_turn", }); expect(explicit.reasoning?.context).toBe("current_turn"); - - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning?.context).toBe("all_turns"); }); - it("defaults reasoning.context to all_turns under Responses Lite unless overridden", async () => { - const model = createCodexModel("gpt-5.1-codex"); + it("keeps the all_turns default for the lite transport on supported models", async () => { + const model = createCodexModel("gpt-5.5"); const lite = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", @@ -112,6 +112,39 @@ describe("openai-codex reasoning.context", () => { }); expect(overridden.reasoning?.context).toBe("auto"); }); + + // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns` + // ("Unsupported value: 'all_turns' is not supported with this model"). + it.each([ + "gpt-5.1-codex", + "gpt-5.3-codex", + "gpt-5.3-codex-spark", + ])("omits the all_turns default for pre-5.4 model %s", async modelId => { + const model = createCodexModel(modelId); + + const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); + expect(defaulted.reasoning).toBeDefined(); + expect(defaulted.reasoning?.context).toBeUndefined(); + expect("context" in (defaulted.reasoning ?? {})).toBe(false); + + // A supported override (current_turn/auto) is still honored. + const overridden = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningContext: "current_turn", + }); + expect(overridden.reasoning?.context).toBe("current_turn"); + }); + + it("suppresses an explicit all_turns override on a pre-5.4 model", async () => { + const model = createCodexModel("gpt-5.3-codex-spark"); + + const forced = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningContext: "all_turns", + }); + expect(forced.reasoning).toBeDefined(); + expect(forced.reasoning?.context).toBeUndefined(); + }); }); describe("openai-codex Responses Lite input shaping", () => { diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index c48d91f35..031a77c31 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -13,6 +13,7 @@ import { parseAnthropicModel, parseGlmModel, parseKnownModel, + parseOpenAIModel, semverGte, } from "./classify"; @@ -121,6 +122,22 @@ export const isOpenAIModelId = memo((modelId: string): boolean => { return /(^|\/)(gpt|o1|o3|o4)[-.]/i.test(modelId) || modelId.toLowerCase().includes("openai/"); }); +/** + * OpenAI Codex models that honor `reasoning.context: "all_turns"` (full + * cross-turn reasoning replay). The `reasoning.context` field itself exists for + * the whole gpt-5/o-series family, but the `all_turns` value is only accepted + * from gpt-5.4 onward; earlier ids (`gpt-5.1-codex`, `gpt-5.3-codex`, and + * `gpt-5.3-codex-spark`) reject it with + * `Unsupported value: 'all_turns' is not supported with this model`. Version + * floor (not an allowlist) so 5.6/6.x inherit support automatically. Callers + * fall back to omitting `context`, letting the server default to `current_turn`. + */ +export const supportsAllTurnsReasoningContext = memo((modelId: string): boolean => { + const parsed = parseOpenAIModel(bareModelId(modelId)); + if (!parsed) return false; + return semverGte(parsed.version, "5.4"); +}); + /** * Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` / * `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning From b704ac698ace685609f9e6dbb2f69af0ed502b1f Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 06:44:42 +0200 Subject: [PATCH 38/88] chore: update changelogs --- packages/ai/CHANGELOG.md | 27 ++++++++----------- packages/catalog/CHANGELOG.md | 4 +-- packages/coding-agent/CHANGELOG.md | 12 ++++----- .../test/agent-session-retry-cap.test.ts | 4 +++ .../web/search/cli-provider-settings.test.ts | 19 +++++++++---- packages/natives/CHANGELOG.md | 2 +- packages/snapcompact/CHANGELOG.md | 6 ++--- packages/stats/CHANGELOG.md | 2 +- packages/tui/CHANGELOG.md | 2 +- 9 files changed, 42 insertions(+), 36 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index b3e3e89e5..89eb89c73 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,28 +4,23 @@ ### Added -- Added service tier support for Google Gemini and Vertex AI -- Introduced `ServiceTierByFamily` to allow model-specific service tier configurations -- Added Google Vertex AI Interactions API support, sharing one Interactions transport across the direct Google and Vertex providers. Gemini 3+ models use the Interactions API by default — on the official `generativelanguage` endpoint for direct Google, and under ADC/bearer auth for Vertex — with automatic fallback to `:streamGenerateContent` when the model/endpoint can't serve Interactions. Pass `useInteractionsApi: false` to force generateContent. -- Added support for an explicit Vertex bearer access token via `GOOGLE_CLOUD_ACCESS_TOKEN` / `CLOUDSDK_AUTH_ACCESS_TOKEN`, so `gcloud auth print-access-token` can drive Vertex without a full `application-default login`. +- Added service tier support for Google Gemini and Vertex AI, including model-specific service tier configurations via ServiceTierByFamily. +- Added Google Vertex AI Interactions API support for Gemini 3+ models by default, with automatic fallback to :streamGenerateContent and a useInteractionsApi: false option to force standard generation. +- Added support for explicit Vertex bearer access tokens via GOOGLE_CLOUD_ACCESS_TOKEN or CLOUDSDK_AUTH_ACCESS_TOKEN environment variables. ### Changed -- Updated service tier logic to avoid global scopes in favor of per-provider configurations -- Refactored priority request billing to better align with specific provider capabilities -- Updated internal `coerceServiceTierByFamily` helper to facilitate migration from legacy settings -- Changed API-key resolution precedence so an explicit environment variable (e.g. `GEMINI_API_KEY`) overrides a stored/broker-migrated static API key; a deliberate OAuth login still takes precedence over the env var. +- Updated service tier logic to use per-provider configurations instead of global scopes. +- Refactored priority request billing and accounting to better align with specific provider capabilities. +- Updated API key resolution precedence so explicit environment variables (e.g., GEMINI_API_KEY) override stored or broker-migrated static API keys, while deliberate OAuth logins still take highest precedence. ### Fixed -- Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors - -- Fixed safety setting application for Google Vertex AI models -- Ensured Gemini service tier is correctly passed through to the API -- Corrected priority request accounting for supported providers -- Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) -- Healed leaked reasoning fences (` ```thinking ` / ``) live for every provider via a central stream wrapper, splitting them into structured thinking blocks during streaming. -- Fixed Codex requests failing with `Unsupported value: 'all_turns' is not supported with this model`: the `reasoning.context: "all_turns"` default is now gated to gpt-5.4+ Codex models. Older ids (`gpt-5.1-codex`, `gpt-5.3-codex`, `gpt-5.3-codex-spark`) omit `context` so the server applies its `current_turn` default; an explicit `all_turns` override is also suppressed on those models, while `current_turn`/`auto` always pass through. +- Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors. +- Fixed safety setting application for Google Vertex AI models. +- Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation. +- Fixed leaked reasoning fences (such as ```thinking or ) across all providers by splitting them into structured thinking blocks during streaming. +- Fixed Codex requests failing with unsupported all_turns errors on older models (gpt-5.1 and gpt-5.3) by gating the reasoning.context: "all_turns" default to gpt-5.4+ models. ## [16.2.6] - 2026-06-29 diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 6014bf5c0..28a9608f0 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -4,8 +4,8 @@ ### Fixed -- Fixed Kimi K2.7 Code compatibility to avoid disabled thinking and forced tool choice on native Kimi endpoints that require thinking mode. ([#3852](https://github.com/can1357/oh-my-pi/issues/3852)) -- Fixed Cerebras `gemma-4-31b` dynamic discovery to mark the model as image-capable so attached images are serialized as OpenAI Chat Completions `image_url` data URIs. ([#3854](https://github.com/can1357/oh-my-pi/issues/3854)) +- Fixed compatibility with Kimi K2.7 Code on native endpoints to ensure thinking mode is preserved and tool choice is not forced. +- Fixed Cerebras gemma-4-31b dynamic discovery to correctly identify the model as image-capable, enabling proper serialization of attached images. ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b1691cd15..55426fdb6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,12 +2,11 @@ ## [Unreleased] -### Changed +### Breaking Changes -- Replaced the global `serviceTier` setting with `tier.openai`, `tier.anthropic`, and `tier.google` for granular control -- Updated `/fast` to target the service-tier family of the currently selected model -- Updated subagent and advisor tier configuration to use the new per-family setting structure -- Removed `fastModeScope` setting, as per-family scoping is now natively supported via the `tier.*` settings +- Replaced the global `serviceTier` and `fastModeScope` settings with granular, per-family settings (`tier.openai`, `tier.anthropic`, and `tier.google`) to control service tiers, subagents, advisors, and `/fast` mode targets. + +### Changed - Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files. - Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content. @@ -21,8 +20,7 @@ - Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error. - Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized. - Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout. -- Fixed the `Working…` loader staying gone for the rest of a parent turn when a long-running tool (e.g. a `task` subagent) finished inside a transient overlay window (auto-snapcompact, auto-context-full, auto-retry). Those overlays null the working loader on start and the overlay-end handler is the only restorer keyed off the missing loader; if the subagent's `tool_execution_end` lands while the overlay is still active (or its end handler errored before re-arming), the spinner stayed gone until the next turn even though the parent kept streaming. `tool_execution_end` now mirrors the `tool_execution_update` self-heal so the working loader survives a subagent completing inside the overlay window ([#3858](https://github.com/can1357/oh-my-pi/issues/3858)). -- Fixed the working loader disappearing after a subagent (`task`) tool completed while the focused session was still streaming: `tool_execution_end` did not re-arm the loader the way `tool_execution_update` did, so a tool result landing after a transient overlay (auto-compaction / auto-retry) left the UI looking idle ([#3857](https://github.com/can1357/oh-my-pi/issues/3857)). +- Fixed an issue where the `Working...` loader spinner would prematurely disappear or fail to re-arm after a subagent (`task`) tool completed or during transient overlays (such as auto-compaction or auto-retry). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/test/agent-session-retry-cap.test.ts b/packages/coding-agent/test/agent-session-retry-cap.test.ts index 60c54526a..d9b8d1333 100644 --- a/packages/coding-agent/test/agent-session-retry-cap.test.ts +++ b/packages/coding-agent/test/agent-session-retry-cap.test.ts @@ -4,6 +4,7 @@ import { scheduler } from "node:timers/promises"; import { Agent } from "@oh-my-pi/pi-agent-core"; import type { ApiKeyResolveContext, AssistantMessage, ToolCall } from "@oh-my-pi/pi-ai"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; +import * as aiStream from "@oh-my-pi/pi-ai/stream"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -54,6 +55,9 @@ describe("AgentSession retry delay cap", () => { beforeEach(async () => { tempDir = TempDir.createSync("@pi-retry-cap-"); authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + // A live env var now overrides a stored static api_key; these tests rotate stored Anthropic + // credentials, so neutralize env resolution (ignores every provider's ambient env key). + vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); authStorage.setRuntimeApiKey("anthropic", "anthropic-test-key"); modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); }); diff --git a/packages/coding-agent/test/web/search/cli-provider-settings.test.ts b/packages/coding-agent/test/web/search/cli-provider-settings.test.ts index fb64f7c87..c7a3ae18c 100644 --- a/packages/coding-agent/test/web/search/cli-provider-settings.test.ts +++ b/packages/coding-agent/test/web/search/cli-provider-settings.test.ts @@ -1,7 +1,11 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { stripVTControlCharacters } from "node:util"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { setExcludedSearchProviders, setPreferredSearchProvider } from "@oh-my-pi/pi-coding-agent/web/search/provider"; +import { + SEARCH_PROVIDER_ORDER, + setExcludedSearchProviders, + setPreferredSearchProvider, +} from "@oh-my-pi/pi-coding-agent/web/search/provider"; import { __resetDirsFromEnvForTests, setAgentDir, TempDir } from "@oh-my-pi/pi-utils"; import { runSearchCommand } from "../../../src/cli/web-search-cli"; @@ -128,17 +132,22 @@ describe("runSearchCommand provider settings", () => { }); it("treats explicit --provider auto as a one-shot override of the configured preferred provider", async () => { - // Same Tavily preference is configured by `beforeEach`, but no exclusions - // hide Jina here, so the auto chain order (Jina before Tavily) decides. + // Tavily is the configured preference, but `--provider auto` overrides it and walks the + // chain. Restrict eligibility to Jina + Tavily so an ambient broker/OAuth provider + // (gemini, anthropic, codex, perplexity…) can't win on a dev machine; the chain order + // (Jina before Tavily) still decides between the two. const currentTempDir = tempAgentDir; if (!currentTempDir) throw new Error("tempAgentDir missing"); + // Drive the exclusion through settings too — Settings.init re-applies + // `providers.webSearchExclude`, overwriting a bare setExcludedSearchProviders() call. + const onlyJinaTavily = SEARCH_PROVIDER_ORDER.filter(id => id !== "jina" && id !== "tavily"); resetSettingsForTest(); setPreferredSearchProvider("auto"); - setExcludedSearchProviders([]); + setExcludedSearchProviders(onlyJinaTavily); await Settings.init({ inMemory: true, cwd: currentTempDir.path(), - overrides: { "providers.webSearch": "tavily" }, + overrides: { "providers.webSearch": "tavily", "providers.webSearchExclude": onlyJinaTavily }, }); vi.spyOn(globalThis, "fetch").mockImplementation(makeFetchMock()); diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index e2b7abe82..c17560987 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -5,7 +5,7 @@ ### Added - Added embedded Silver TrueType font rendering support to `renderSnapcompactPng`, featuring automatic per-glyph fallback for missing bitmap characters and anti-aliased scaling for East Asian wide code points. -- Added `snapcompactSupportedChars` to check font capability for specific characters. +- Added the `snapcompactSupportedChars` function to check font capability for specific characters. ## [16.2.5] - 2026-06-28 diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index fa3a517e9..a7d33af7e 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -9,9 +9,9 @@ ### Changed -- Improved text normalization for non-ASCII text: semantic emojis fold to ASCII labels (e.g., `[OK]`, `[WARN]`, `[FAIL]`), decorative emojis are dropped, box-drawing/compatibility symbols fold to ASCII skeletons, and Unicode text is preserved when supported by the selected font or the embedded Silver fallback. -- Updated bitmap shapes to draw missing glyphs per-character using the embedded Silver TrueType fallback instead of rendering blanks or switching entire snippets, with support for East Asian wide characters across two grid cells. -- Updated text wrapping, pagination, and provider shape geometries to account for wide character footprints and updated X.org 8x13 font metrics (11px/22px pitches). +- Improved non-ASCII text normalization by folding semantic emojis to ASCII labels (e.g., `[OK]`, `[WARN]`), dropping decorative emojis, and folding box-drawing symbols to ASCII skeletons. +- Enhanced missing glyph rendering to use the embedded Silver TrueType fallback per-character, including support for East Asian wide characters across two grid cells. +- Updated text wrapping, pagination, and provider shape geometries to support wide character footprints and updated X.org 8x13 font metrics. ## [16.1.23] - 2026-06-26 diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index fcfe5d6dc..e131c8cf5 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Improved premium request calculation logic to account for specific model families +- Improved premium request calculation accuracy by correctly accounting for specific model families. ## [16.2.6] - 2026-06-29 diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index cf0d338cb..5b7affbe3 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `StdinBuffer` swallowing a fast double-Esc that arrived as one `"\x1b\x1b"` chunk: `parseKey` returns `undefined` for the combined chunk, so the editor's double-escape gesture and any single-Esc handler the second press should have hit never fired. The buffer now splits a bare `"\x1b\x1b"` into two ESC events only when no follower arrives in the disambiguation window; when a follower arrives, the second ESC stays attached so legacy Alt chords like `"\x1bd"` and meta-CSI/SS3 chords like `"\x1b\x1b[A"` still emit as parseable sequences ([#3857](https://github.com/can1357/oh-my-pi/issues/3857)). +- Fixed an issue where a fast double-Escape keypress was swallowed and ignored, preventing double-escape gestures and subsequent Escape key handlers from firing. ## [16.2.3] - 2026-06-28 From 467f46b1903649efa9b6197afadfeffd32627f07 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 07:01:32 +0200 Subject: [PATCH 39/88] perf(scripts/session-stats): added composite index on tool call timestamp and name - Added a composite index `ss_tc_ts_tool` on the `timestamp` and `tool_name` columns of the `ss_tool_calls` table to optimize query performance. --- scripts/session-stats/sync.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/session-stats/sync.py b/scripts/session-stats/sync.py index 026373ec8..d50132939 100644 --- a/scripts/session-stats/sync.py +++ b/scripts/session-stats/sync.py @@ -96,6 +96,7 @@ CREATE TABLE IF NOT EXISTS ss_tool_calls ( UNIQUE(session_file, call_id, seq) ); CREATE INDEX IF NOT EXISTS ss_tc_tool_ts ON ss_tool_calls(tool_name, timestamp); +CREATE INDEX IF NOT EXISTS ss_tc_ts_tool ON ss_tool_calls(timestamp, tool_name); CREATE INDEX IF NOT EXISTS ss_tc_sess_seq ON ss_tool_calls(session_file, seq); CREATE TABLE IF NOT EXISTS ss_tool_results ( From 09061ed801fc1a7a80f81608c5c83ea713762472 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 07:02:12 +0200 Subject: [PATCH 40/88] feat(coding-agent/web): aligned duckduckgo search requests with native browser behavior - Updated the default browser User-Agent string to emulate a modern version of Chrome. - Added typical browser headers to the outgoing fetch request, including Sec-Ch-Ua, Sec-Fetch flags, and Referer. - Added a blank "b" parameter to the form body to match native DuckDuckGo HTML search behavior. --- packages/coding-agent/CHANGELOG.md | 2 ++ .../src/web/search/providers/duckduckgo.ts | 20 ++++++++++++++++--- .../test/tools/web-search-duckduckgo.test.ts | 5 +++++ 3 files changed, 24 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 55426fdb6..9a7dc7199 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,8 @@ ### Fixed +- Improved reliability of DuckDuckGo web searches by updating browser request headers and parameters + - Fixed an issue where CJK (Chinese, Japanese, Korean) history could become unrenderable during repeated context compactions. - Fixed a memory exhaustion bug in the TUI when using `/resume` on large previous sessions. - Fixed an issue where the `irc` inbox missed messages that arrived while the recipient agent was already running. diff --git a/packages/coding-agent/src/web/search/providers/duckduckgo.ts b/packages/coding-agent/src/web/search/providers/duckduckgo.ts index 078eddb20..15c86ea45 100644 --- a/packages/coding-agent/src/web/search/providers/duckduckgo.ts +++ b/packages/coding-agent/src/web/search/providers/duckduckgo.ts @@ -35,7 +35,7 @@ const RECENCY_TO_DDG_DF: Record, string> = * the orchestrator can fall through to the next provider with context. */ const BROWSER_USER_AGENT = - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"; + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36"; interface ParsedResult { title: string; @@ -130,15 +130,29 @@ async function callDuckDuckGoHtml(params: SearchParams): Promise { const form = new URLSearchParams({ q: params.query, kl: "us-en" }); const df = params.recency ? RECENCY_TO_DDG_DF[params.recency] : undefined; if (df) form.set("df", df); + // Add b: "" parameter as specified in the browser fetch template to match real browser form submission + form.set("b", ""); const response = await (params.fetch ?? fetch)(DUCKDUCKGO_HTML_URL, { method: "POST", body: form.toString(), headers: { + Accept: + "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7", + "Accept-Language": "en,en-US;q=0.9", + "Cache-Control": "max-age=0", "Content-Type": "application/x-www-form-urlencoded", + Priority: "u=0, i", + "Sec-Ch-Ua": '"Google Chrome";v="149", "Chromium";v="149", "Not)A;Brand";v="24"', + "Sec-Ch-Ua-Mobile": "?0", + "Sec-Ch-Ua-Platform": '"macOS"', + "Sec-Fetch-Dest": "document", + "Sec-Fetch-Mode": "navigate", + "Sec-Fetch-Site": "same-origin", + "Sec-Fetch-User": "?1", + "Upgrade-Insecure-Requests": "1", "User-Agent": BROWSER_USER_AGENT, - Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", - "Accept-Language": "en-US,en;q=0.5", + Referer: "https://html.duckduckgo.com/", }, signal: withHardTimeout(params.signal), }); diff --git a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts index 05882600d..4da54bddd 100644 --- a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts +++ b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts @@ -68,10 +68,15 @@ describe("DuckDuckGo web search provider", () => { const form = new URLSearchParams(capturedInit?.body as string); expect(form.get("q")).toBe("how to fix bug in code"); expect(form.get("kl")).toBe("us-en"); + expect(form.get("b")).toBe(""); expect(form.get("df")).toBe("w"); const headers = capturedInit?.headers as Record; expect(headers["Content-Type"]).toBe("application/x-www-form-urlencoded"); expect(headers["User-Agent"]).toContain("Mozilla/5.0"); + expect(headers["Referer"]).toBe("https://html.duckduckgo.com/"); + expect(headers["Accept-Language"]).toContain("en"); + expect(headers["Sec-Fetch-Mode"]).toBe("navigate"); + expect(headers["Sec-Ch-Ua"]).toContain("Chromium"); }); it("omits the df form param when no recency is requested", async () => { From 38250ce88b5bbd4f58068e14b7fa61164ad2654c Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 30 Jun 2026 07:20:23 +0200 Subject: [PATCH 41/88] chore: bump version to 16.2.7 --- Cargo.lock | 16 +++---- Cargo.toml | 2 +- bun.lock | 60 +++++++++++++-------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 24 +++++------ packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 + packages/ai/package.json | 2 +- packages/catalog/CHANGELOG.md | 2 + packages/catalog/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 3 +- packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/CHANGELOG.md | 2 + packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/CHANGELOG.md | 2 + packages/snapcompact/package.json | 2 +- packages/stats/CHANGELOG.md | 2 + packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 + packages/tui/package.json | 2 +- packages/utils/CHANGELOG.md | 2 + packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 28 files changed, 81 insertions(+), 70 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index bdf91c7ba..1157ece8c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2064,9 +2064,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.29" +version = "0.2.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34f877a98676d2fb664698d74cc6a51ce6c484ce8c770f05d0108ec9090aeb46" +checksum = "ccfe6121cbe750cf81efa362d85c0bde7ea298ec43092d3a193baca59cdbd634" dependencies = [ "defmt", "jiff-static", @@ -2091,9 +2091,9 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.29" +version = "0.2.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0666b5ab5ecaca213fc2a85b8c0083d9004e84ee2d5f9a7e0017aaf50986f25f" +checksum = "e165e897f662d428f3cd3828a919dbe067c2d42bb1031eede74ef9d27ecdedd2" dependencies = [ "proc-macro2", "quote", @@ -2911,7 +2911,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.2.6" +version = "16.2.7" dependencies = [ "anyhow", "ast-grep-core", @@ -2981,7 +2981,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.2.6" +version = "16.2.7" dependencies = [ "async-trait", "libc", @@ -2993,7 +2993,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.2.6" +version = "16.2.7" dependencies = [ "anyhow", "arboard", @@ -3043,7 +3043,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.2.6" +version = "16.2.7" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index e4033c8f8..2f93a6ac7 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.2.6" +version = "16.2.7" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index d9c1512f9..7c657d35d 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.2.6", + "version": "16.2.7", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.2.6", + "version": "16.2.7", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.2.6", + "version": "16.2.7", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.2.6", + "version": "16.2.7", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.2.6", + "version": "16.2.7", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.2.6", + "version": "16.2.7", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.2.6", + "version": "16.2.7", "devDependencies": { "@types/bun": "catalog:", }, @@ -337,18 +337,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.2.6", - "@oh-my-pi/omp-stats": "16.2.6", - "@oh-my-pi/pi-agent-core": "16.2.6", - "@oh-my-pi/pi-ai": "16.2.6", - "@oh-my-pi/pi-catalog": "16.2.6", - "@oh-my-pi/pi-coding-agent": "16.2.6", - "@oh-my-pi/pi-mnemopi": "16.2.6", - "@oh-my-pi/pi-natives": "16.2.6", - "@oh-my-pi/pi-tui": "16.2.6", - "@oh-my-pi/pi-utils": "16.2.6", - "@oh-my-pi/pi-wire": "16.2.6", - "@oh-my-pi/snapcompact": "16.2.6", + "@oh-my-pi/hashline": "16.2.7", + "@oh-my-pi/omp-stats": "16.2.7", + "@oh-my-pi/pi-agent-core": "16.2.7", + "@oh-my-pi/pi-ai": "16.2.7", + "@oh-my-pi/pi-catalog": "16.2.7", + "@oh-my-pi/pi-coding-agent": "16.2.7", + "@oh-my-pi/pi-mnemopi": "16.2.7", + "@oh-my-pi/pi-natives": "16.2.7", + "@oh-my-pi/pi-tui": "16.2.7", + "@oh-my-pi/pi-utils": "16.2.7", + "@oh-my-pi/pi-wire": "16.2.7", + "@oh-my-pi/snapcompact": "16.2.7", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -1042,7 +1042,7 @@ "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], - "electron-to-chromium": ["electron-to-chromium@1.5.379", "", {}, "sha512-v/qV5aV5EUA2pGilzUCq5/eyOloZAqDZBu9UMBIzgPpLlprjSR6zswsWBTv0KpqxLGUAZEwhO95ZCt7srymNVA=="], + "electron-to-chromium": ["electron-to-chromium@1.5.380", "", {}, "sha512-W6d5AbuEoRayO447cqrg6lKJIlscgRnnxOZl/08kfV71BQDoEBC7Wwis68z87LjyK6f4kWyTaubuDbhHKrZkbA=="], "emnapi": ["emnapi@1.11.1", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-kSRjhIcxjMFsBqk7ORvoc9aA5SBKDmecrtF5RMcmOTao0kD/zamaxsuTxMI8C1//wGUuvE7a+19pCE7AEhGVnA=="], @@ -1146,7 +1146,7 @@ "js-tokens": ["js-tokens@4.0.0", "", {}, "sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ=="], - "js-yaml": ["js-yaml@4.2.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw=="], + "js-yaml": ["js-yaml@4.3.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q=="], "jsesc": ["jsesc@3.1.0", "", { "bin": { "jsesc": "bin/jsesc" } }, "sha512-/sM3dO2FOzXjKQhJuo0Q173wf2KOo8t4I8vHy6lF9poUp7bKT0/NHE8fPX23PwfhnykfqnC2xRxOnVw5XuGIaA=="], @@ -1370,7 +1370,7 @@ "string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], - "string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="], + "string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], "strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="], @@ -1534,8 +1534,6 @@ "roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="], - "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], - "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], "@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], @@ -1556,8 +1554,6 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.17", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-wPEBwzapC+2PaTYPH6e2L+cNOEE227S47wUYFqlegcs8zlLLmeb9Fcff1HVZY4Fwku/1Eyv38n7GYwB2aaS71g=="], - "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], - "@huggingface/transformers/onnxruntime-node/global-agent/matcher": ["matcher@3.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng=="], "@huggingface/transformers/onnxruntime-node/global-agent/serialize-error": ["serialize-error@7.0.1", "", { "dependencies": { "type-fest": "^0.13.1" } }, "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index c1b4f9742..a3b07d7dd 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -173,7 +173,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_2_6")] +#[napi(js_name = "__piNativesV16_2_7")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index dd1b7c615..ffd35c3ff 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.2.6", - "@oh-my-pi/omp-stats": "16.2.6", - "@oh-my-pi/pi-agent-core": "16.2.6", - "@oh-my-pi/pi-ai": "16.2.6", - "@oh-my-pi/pi-catalog": "16.2.6", - "@oh-my-pi/pi-coding-agent": "16.2.6", - "@oh-my-pi/pi-mnemopi": "16.2.6", - "@oh-my-pi/pi-natives": "16.2.6", - "@oh-my-pi/pi-tui": "16.2.6", - "@oh-my-pi/pi-utils": "16.2.6", - "@oh-my-pi/pi-wire": "16.2.6", - "@oh-my-pi/snapcompact": "16.2.6", + "@oh-my-pi/hashline": "16.2.7", + "@oh-my-pi/omp-stats": "16.2.7", + "@oh-my-pi/pi-agent-core": "16.2.7", + "@oh-my-pi/pi-ai": "16.2.7", + "@oh-my-pi/pi-catalog": "16.2.7", + "@oh-my-pi/pi-coding-agent": "16.2.7", + "@oh-my-pi/pi-mnemopi": "16.2.7", + "@oh-my-pi/pi-natives": "16.2.7", + "@oh-my-pi/pi-tui": "16.2.7", + "@oh-my-pi/pi-utils": "16.2.7", + "@oh-my-pi/pi-wire": "16.2.7", + "@oh-my-pi/snapcompact": "16.2.7", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index bb031de05..c1fd1bc84 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.2.6", + "version": "16.2.7", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 89eb89c73..cc98662b5 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Added - Added service tier support for Google Gemini and Vertex AI, including model-specific service tier configurations via ServiceTierByFamily. diff --git a/packages/ai/package.json b/packages/ai/package.json index 3ccd37c1f..e5c6a44ac 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.2.6", + "version": "16.2.7", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 28a9608f0..093e78428 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Fixed - Fixed compatibility with Kimi K2.7 Code on native endpoints to ensure thinking mode is preserved and tool choice is not forced. diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 9f09ce14e..7f57b414b 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.2.6", + "version": "16.2.7", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9a7dc7199..c44e8b2da 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Breaking Changes - Replaced the global `serviceTier` and `fastModeScope` settings with granular, per-family settings (`tier.openai`, `tier.anthropic`, and `tier.google`) to control service tiers, subagents, advisors, and `/fast` mode targets. @@ -14,7 +16,6 @@ ### Fixed - Improved reliability of DuckDuckGo web searches by updating browser request headers and parameters - - Fixed an issue where CJK (Chinese, Japanese, Korean) history could become unrenderable during repeated context compactions. - Fixed a memory exhaustion bug in the TUI when using `/resume` on large previous sessions. - Fixed an issue where the `irc` inbox missed messages that arrived while the recipient agent was already running. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index af8c5dd51..ede897870 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.2.6", + "version": "16.2.7", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index c4dd33605..de14ea732 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.2.6", + "version": "16.2.7", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 49c5f8bf3..0957b0846 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.2.6", + "version": "16.2.7", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index c17560987..2449f09f1 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Added - Added embedded Silver TrueType font rendering support to `renderSnapcompactPng`, featuring automatic per-glyph fallback for missing bitmap characters and anti-aliased scaling for East Asian wide code points. diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index a67fbf4ba..d6117c288 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -162,7 +162,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_2_6(): void +export declare function __piNativesV16_2_7(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index b74be402b..e9ab5a35f 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_2_6 = nativeBindings.__piNativesV16_2_6; +export const __piNativesV16_2_7 = nativeBindings.__piNativesV16_2_7; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 12b32e85f..8c17af6e6 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.2.6", + "version": "16.2.7", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index a7d33af7e..09fced414 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Added - Added the `silver16-bw` shape backed by an embedded Silver TrueType font to support CJK and other non-Latin text. diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index fc5ee9d14..67e20005a 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.2.6", + "version": "16.2.7", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index e131c8cf5..1cbaed0ca 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Fixed - Improved premium request calculation accuracy by correctly accounting for specific model families. diff --git a/packages/stats/package.json b/packages/stats/package.json index 57938cf0a..c050f4284 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.2.6", + "version": "16.2.7", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 636c33654..e3ad04d19 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.2.6", + "version": "16.2.7", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5b7affbe3..c4fb198e2 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Fixed - Fixed an issue where a fast double-Escape keypress was swallowed and ignored, preventing double-escape gestures and subsequent Escape key handlers from firing. diff --git a/packages/tui/package.json b/packages/tui/package.json index be8420c28..8851a1a51 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.2.6", + "version": "16.2.7", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 0b05df86e..30553986d 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.2.7] - 2026-06-30 + ### Added - Added a utility to detect binary files based on content sniffing. diff --git a/packages/utils/package.json b/packages/utils/package.json index fb12a18e0..a36e723e6 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.2.6", + "version": "16.2.7", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index 4d29741f6..063933259 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.2.6", + "version": "16.2.7", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From 88308f105f5f087971ba1a42a92d7d60b6b1c2ba Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:36:14 +0000 Subject: [PATCH 42/88] fix(compaction): capped snapcompact frame payloads Bounded persisted snapcompact image archives by base64 byte size so large sessions stop re-sending multi-megabyte frame walls on every provider request. Auto snapcompact now falls back to context-full summaries when rendered frame payloads exceed the byte budget, and legacy oversized archives omit over-budget frames during LLM context rebuilds. Fixes #3792 --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/session/agent-session.ts | 42 ++++++++- .../src/session/session-context.ts | 15 ++- .../agent-session-snapcompact-budget.test.ts | 1 + .../session-manager/build-context.test.ts | 39 ++++++++ packages/snapcompact/CHANGELOG.md | 4 + packages/snapcompact/src/snapcompact.ts | 94 +++++++++++++++++-- 7 files changed, 185 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d231182bd..b44911364 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763)) +- Fixed long snapcompact sessions re-sending multi-megabyte standing image archives on every provider request by enforcing a per-request frame byte budget, letting auto-compaction fall back to context-full summaries when snapcompact output is too large, and omitting legacy over-budget frames from rebuilt LLM contexts. ([#3792](https://github.com/can1357/oh-my-pi/issues/3792)) ## [16.2.5] - 2026-06-28 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index c7bef0d03..58ff10dd9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -8914,6 +8914,22 @@ export class AgentSession { shape: snapcompact.resolveShape(this.model, this.settings.get("snapcompact.shape")), maxFrames, }); + const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); + if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { + logger.warn("Snapcompact exceeded the per-request frame payload budget", { + model: this.model?.id, + framePayloadBytes, + budget: snapcompact.FRAME_DATA_BYTES_BUDGET, + }); + this.emitNotice( + "warning", + "snapcompact produced too much standing image payload. No LLM fallback was attempted.", + "compaction", + ); + throw new Error( + "snapcompact cannot run locally: standing image payload exceeds the per-request budget.", + ); + } const ctxWindow = this.model?.contextWindow ?? 0; const budget = ctxWindow > 0 @@ -10915,7 +10931,16 @@ export class AgentSession { const capReserve = textEdgeTokens + SUMMARY_TEMPLATE_TOKENS; const frameBudget = totalBudget - baseTokens - capReserve; if (frameBudget < snapcompact.FRAME_TOKEN_ESTIMATE) return 1; - return Math.min(Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), snapcompact.MAX_FRAMES_DEFAULT); + return Math.min( + Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), + snapcompact.MAX_FRAMES_DEFAULT, + snapcompact.maxFramesForDataBudget(), + ); + } + + #snapcompactFramePayloadBytes(result: snapcompact.CompactionResult): number { + const archive = snapcompact.getPreservedArchive(result.preserveData); + return archive ? snapcompact.frameDataBytes(archive.frames) : 0; } /** @@ -10928,7 +10953,9 @@ export class AgentSession { */ #projectSnapcompactContextTokens(preparation: CompactionPreparation, result: snapcompact.CompactionResult): number { const archive = snapcompact.getPreservedArchive(result.preserveData); - const blocks = archive ? snapcompact.historyBlocks(archive) : undefined; + const blocks = archive + ? snapcompact.historyBlocks(archive, { maxFrameDataBytes: snapcompact.FRAME_DATA_BYTES_BUDGET }) + : undefined; const summaryMessage = createCompactionSummaryMessage( result.summary, result.tokensBefore, @@ -11286,6 +11313,17 @@ export class AgentSession { shape: snapcompact.resolveShape(this.model, this.settings.get("snapcompact.shape")), maxFrames, }); + const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); + if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { + logger.warn("Snapcompact exceeded the per-request frame payload budget", { + model: this.model?.id, + framePayloadBytes, + budget: snapcompact.FRAME_DATA_BYTES_BUDGET, + }); + snapcompactBlocker = + "snapcompact produced too much standing image payload; using context-full auto-compaction instead."; + snapcompactResult = undefined; + } if (snapcompactResult) { const ctxWindow = this.model?.contextWindow ?? 0; const budget = diff --git a/packages/coding-agent/src/session/session-context.ts b/packages/coding-agent/src/session/session-context.ts index 787da464a..2b83af9ba 100644 --- a/packages/coding-agent/src/session/session-context.ts +++ b/packages/coding-agent/src/session/session-context.ts @@ -77,6 +77,17 @@ export interface BuildSessionContextOptions { * If leafId is provided, walks from that entry to root. * Handles compaction and branch summaries along the path. */ +function snapcompactHistoryBlocksForContext( + archive: snapcompact.Archive | undefined, + options: BuildSessionContextOptions | undefined, +) { + if (!archive) return undefined; + return snapcompact.historyBlocks( + archive, + options?.transcript ? undefined : { maxFrameDataBytes: snapcompact.FRAME_DATA_BYTES_BUDGET }, + ); +} + export function buildSessionContext( entries: SessionEntry[], leafId?: string | null, @@ -273,7 +284,7 @@ export function buildSessionContext( entry.shortSummary, undefined, undefined, - snapcompactArchive ? snapcompact.historyBlocks(snapcompactArchive) : undefined, + snapcompactHistoryBlocksForContext(snapcompactArchive, options), ), ); } else { @@ -307,7 +318,7 @@ export function buildSessionContext( compaction.shortSummary, providerPayload, undefined, - snapcompactArchive ? snapcompact.historyBlocks(snapcompactArchive) : undefined, + snapcompactHistoryBlocksForContext(snapcompactArchive, options), ), ); diff --git a/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts b/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts index 6080241f8..249ebeae1 100644 --- a/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts +++ b/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts @@ -163,6 +163,7 @@ describe("AgentSession snapcompact frame-budget sizing", () => { const maxFrames = opts?.maxFrames; expect(maxFrames).toBeDefined(); expect(maxFrames).toBeLessThan(snapcompact.MAX_FRAMES_DEFAULT); + expect(maxFrames).toBeLessThanOrEqual(snapcompact.maxFramesForDataBudget()); expect(maxFrames).toBeGreaterThan(0); // Verify the FULL projection — base (non-message + kept-recent) + diff --git a/packages/coding-agent/test/session-manager/build-context.test.ts b/packages/coding-agent/test/session-manager/build-context.test.ts index bc3d94ffb..f703f4138 100644 --- a/packages/coding-agent/test/session-manager/build-context.test.ts +++ b/packages/coding-agent/test/session-manager/build-context.test.ts @@ -8,6 +8,7 @@ import type { SessionMessageEntry, ThinkingLevelChangeEntry, } from "@oh-my-pi/pi-coding-agent/session/session-entries"; +import * as snapcompact from "@oh-my-pi/snapcompact"; function msg(id: string, parentId: string | null, role: "user" | "assistant", text: string): SessionMessageEntry { const base = { type: "message" as const, id, parentId, timestamp: "2025-01-01T00:00:00Z" }; @@ -230,6 +231,44 @@ describe("buildSessionContext", () => { expect((ctx.messages[1] as { content: string }).content).toBe("after compact"); }); + it("caps snapcompact frame payload in LLM context but preserves transcript frames", () => { + const largeFrame = "a".repeat(Math.ceil(snapcompact.FRAME_DATA_BYTES_BUDGET / 2) + 1); + const compacted: CompactionEntry = { + ...compaction("3", "2", "Snapcompact summary", "1"), + preserveData: { + [snapcompact.PRESERVE_KEY]: { + frames: [ + { data: largeFrame, mimeType: "image/png", cols: 10, rows: 10, chars: 10 }, + { data: largeFrame, mimeType: "image/png", cols: 10, rows: 10, chars: 10 }, + ], + totalChars: 20, + truncatedChars: 0, + textHead: "old edge", + textTail: "new edge", + }, + }, + }; + const entries: SessionEntry[] = [ + msg("1", null, "user", "first"), + msg("2", "1", "assistant", "response"), + compacted, + msg("4", "3", "user", "after compact"), + ]; + + const llmContext = buildSessionContext(entries); + const summary = llmContext.messages[0]; + if (summary?.role !== "compactionSummary") throw new Error("Expected LLM compaction summary"); + expect(summary.blocks?.filter(block => block.type === "image")).toHaveLength(1); + expect( + summary.blocks?.some(block => block.type === "text" && block.text.includes("image middle omitted")), + ).toBe(true); + + const transcript = buildSessionContext(entries, undefined, undefined, { transcript: true }); + const transcriptSummary = transcript.messages[2]; + if (transcriptSummary?.role !== "compactionSummary") throw new Error("Expected transcript compaction summary"); + expect(transcriptSummary.blocks?.filter(block => block.type === "image")).toHaveLength(2); + }); + it("multiple compactions uses latest", () => { const entries: SessionEntry[] = [ msg("1", null, "user", "a"), diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index e3d737515..ab9012c7b 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed large snapcompact archives being reconstructed into unbounded per-request image payloads by adding a frame base64 byte budget and omitting over-budget archive frames from prompt blocks. ([#3792](https://github.com/can1357/oh-my-pi/issues/3792)) + ## [16.1.23] - 2026-06-26 ### Added diff --git a/packages/snapcompact/src/snapcompact.ts b/packages/snapcompact/src/snapcompact.ts index ba84dd266..b95f71e46 100644 --- a/packages/snapcompact/src/snapcompact.ts +++ b/packages/snapcompact/src/snapcompact.ts @@ -404,6 +404,29 @@ export const HQ_EDGE_FRAMES = 3; * undercounting a high-res archive at the raised {@link MAX_FRAMES_DEFAULT}. */ export const FRAME_TOKEN_ESTIMATE = 5024; +/** Conservative upper bound for one persisted frame's base64 payload. The + * measured high-res Anthropic `8x13`/`11on16` PNG frames sit around 159 KB; + * 170 KB leaves margin for denser glyph pages without permitting multi-MB + * standing request bodies at large context windows. */ +export const FRAME_DATA_BYTES_ESTIMATE = 170_000; + +/** Maximum snapcompact image base64 carried in every rebuilt provider request. + * Above this, provider backends can accept the HTTP body but fail mid-stream + * with opaque 5xx errors. Keep this independent from visual-token budgeting: + * a 1M-token model can afford 70 images on paper, but not the resulting + * ~11 MB JSON payload on every turn. */ +export const FRAME_DATA_BYTES_BUDGET = 3_000_000; + +/** Frame-count cap implied by {@link FRAME_DATA_BYTES_BUDGET}. */ +export function maxFramesForDataBudget(maxFrameDataBytes: number = FRAME_DATA_BYTES_BUDGET): number { + return Math.max(1, Math.floor(maxFrameDataBytes / FRAME_DATA_BYTES_ESTIMATE)); +} + +/** Base64 byte length for persisted snapcompact frames. */ +export function frameDataBytes(frames: readonly Pick[]): number { + return frames.reduce((sum, frame) => sum + frame.data.length, 0); +} + /** * Per-request image-count budgets by provider id. These cap how many images an * entire request may carry (archive/system-prompt/tool-result imaging combined). @@ -1313,6 +1336,51 @@ export function archiveSourceText(archive: Archive): string | undefined { return text.length > 0 ? toPlainText(text) : undefined; } +/** Options for reconstructing a persisted snapcompact archive into prompt blocks. */ +export interface HistoryBlockOptions { + /** Hard cap on image base64 bytes attached to one rebuilt provider request. */ + maxFrameDataBytes?: number; +} + +function formatFrameDataBytes(bytes: number): string { + if (bytes >= 1_000_000) return `${(bytes / 1_000_000).toFixed(1)} MB`; + if (bytes >= 1_000) return `${(bytes / 1_000).toFixed(1)} KB`; + return `${bytes} B`; +} + +function imagesWithinBudget( + archive: Archive, + maxFrameDataBytes: number | undefined, +): { images: ImageContent[]; omittedFrames: number; omittedBytes: number } { + if (maxFrameDataBytes === undefined) { + return { images: images(archive), omittedFrames: 0, omittedBytes: 0 }; + } + + let usedBytes = 0; + let omittedFrames = 0; + let omittedBytes = 0; + const kept: Frame[] = []; + for (const frame of archive.frames) { + const bytes = frame.data.length; + if (usedBytes + bytes > maxFrameDataBytes) { + omittedFrames++; + omittedBytes += bytes; + continue; + } + usedBytes += bytes; + kept.push(frame); + } + return { images: images({ ...archive, frames: kept }), omittedFrames, omittedBytes }; +} + +function omittedFrameNotice(omittedFrames: number, omittedBytes: number): string { + return [ + "-------------- snapcompact image middle omitted", + `${omittedFrames.toLocaleString()} archived image frame${omittedFrames === 1 ? "" : "s"} (${formatFrameDataBytes(omittedBytes)} base64) exceeded the per-request snapcompact payload budget. The compacted summary and visible text edges remain available.`, + "--------------", + ].join("\n"); +} + /** Convert archive frames into LLM image blocks (oldest first). */ export function images(archive: Archive): ImageContent[] { return archive.frames.map(frame => ({ @@ -1326,23 +1394,35 @@ export function images(archive: Archive): ImageContent[] { * the oldest text region, the imaged middle, then the newest text region. * Runtime-only; reconstructed from {@link Archive} on each context rebuild * instead of persisted on the session entry. */ -export function historyBlocks(archive: Archive): (TextContent | ImageContent)[] { +export function historyBlocks(archive: Archive, options: HistoryBlockOptions = {}): (TextContent | ImageContent)[] { const blocks: (TextContent | ImageContent)[] = []; - const hasImages = archive.frames.length > 0; + const budgeted = imagesWithinBudget(archive, options.maxFrameDataBytes); + const hasImages = budgeted.images.length > 0; + const hasOmittedImages = budgeted.omittedFrames > 0; if (archive.textHead) { - const suffix = hasImages ? "\n-------------- imaged middle below\n" : ""; + const suffix = hasImages + ? "\n-------------- imaged middle below\n" + : hasOmittedImages + ? `\n${omittedFrameNotice(budgeted.omittedFrames, budgeted.omittedBytes)}\n` + : ""; blocks.push({ type: "text", text: toPlainText(archive.textHead) + suffix }); + } else if (hasOmittedImages) { + blocks.push({ type: "text", text: omittedFrameNotice(budgeted.omittedFrames, budgeted.omittedBytes) }); + } + blocks.push(...budgeted.images); + if (hasImages && hasOmittedImages) { + blocks.push({ type: "text", text: omittedFrameNotice(budgeted.omittedFrames, budgeted.omittedBytes) }); } - blocks.push(...images(archive)); if (archive.textTail) { const prefix = hasImages ? "-------------- imaged middle above\n" - : archive.truncatedChars > 0 + : archive.truncatedChars > 0 || hasOmittedImages ? "\n-------------- middle history omitted above\n" : ""; const tail = prefix + toPlainText(archive.textTail); - if (blocks.length > 0 && blocks[blocks.length - 1]?.type === "text") { - (blocks[blocks.length - 1] as TextContent).text += tail; + const lastBlock = blocks[blocks.length - 1]; + if (lastBlock?.type === "text") { + lastBlock.text += tail; } else { blocks.push({ type: "text", text: tail }); } From 156dfd8467a763e8b037670c0bfcc25615e8b456 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:43:17 +0000 Subject: [PATCH 43/88] fix(compaction): capped unknown-window snapcompact frames Apply the snapcompact frame byte-budget cap even when the active model has no known context window, avoiding 80-frame archives on custom vision models. --- .../coding-agent/src/session/agent-session.ts | 2 +- .../agent-session-snapcompact-budget.test.ts | 37 +++++++++++++++++++ 2 files changed, 38 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 58ff10dd9..e93f662d4 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -10892,7 +10892,7 @@ export class AgentSession { */ #computeSnapcompactMaxFrames(preparation: CompactionPreparation, settings: CompactionSettings): number { const ctxWindow = this.model?.contextWindow ?? 0; - if (ctxWindow <= 0) return snapcompact.MAX_FRAMES_DEFAULT; + if (ctxWindow <= 0) return Math.min(snapcompact.MAX_FRAMES_DEFAULT, snapcompact.maxFramesForDataBudget()); const reserve = effectiveReserveTokens(ctxWindow, settings); let baseTokens = computeNonMessageTokens(this); for (const message of preparation.recentMessages) { diff --git a/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts b/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts index 249ebeae1..aa92fe7b5 100644 --- a/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts +++ b/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts @@ -242,4 +242,41 @@ describe("AgentSession snapcompact frame-budget sizing", () => { // text-only `planArchive` path makes this case recoverable. expect(opts?.maxFrames).toBe(1); }); + + it("applies the frame byte cap when the model context window is unknown", async () => { + const model = session.model; + if (!model) throw new Error("Expected model"); + await session.dispose(); + const unknownWindowModel = { ...model, contextWindow: 0 }; + session = new AgentSession({ + agent: new Agent({ + initialState: { model: unknownWindowModel, systemPrompt: ["Test"], tools: [], messages: [] }, + }), + sessionManager, + settings: Settings.isolated({ + "compaction.strategy": "snapcompact", + "compaction.autoContinue": false, + "compaction.keepRecentTokens": 4000, + }), + modelRegistry, + }); + + const branchEntries = sessionManager.getBranch(); + const lastEntry = branchEntries[branchEntries.length - 1]; + if (!lastEntry?.id) throw new Error("Expected branch entry with id"); + const compactSpy = vi.spyOn(snapcompact, "compact").mockResolvedValue({ + summary: "stubbed snapcompact", + shortSummary: "stub", + firstKeptEntryId: lastEntry.id, + tokensBefore: 100_000, + details: { readFiles: [], modifiedFiles: [] }, + preserveData: { + snapcompact: { frames: [], totalChars: 0, truncatedChars: 0 }, + }, + }); + + await session.compact(undefined, { mode: "snapcompact" }); + + expect(compactSpy.mock.calls[0]?.[1]?.maxFrames).toBe(snapcompact.maxFramesForDataBudget()); + }); }); From d7c64728245cd0a0ae3cae6900ee187be33f7cb7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:48:37 +0000 Subject: [PATCH 44/88] fix(edit): preserved hashline utf-8 bom Added binary BOM detection to the hashline patcher and filesystem adapters so edits restore BOM bytes when text decoding hides U+FEFF. Fixes #3867 --- packages/coding-agent/CHANGELOG.md | 4 +++ .../src/edit/hashline/filesystem.ts | 10 ++++++++ .../coding-agent/test/core/hashline.test.ts | 17 +++++++++++++ packages/hashline/CHANGELOG.md | 4 +++ packages/hashline/src/fs.ts | 12 +++++++++ packages/hashline/src/patcher.ts | 13 +++++++++- packages/hashline/test/patcher.test.ts | 25 +++++++++++++++++++ 7 files changed, 84 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c44e8b2da..e4922c2a0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed hashline edit mode preserving UTF-8 BOM bytes on edited files. ([#3867](https://github.com/can1357/oh-my-pi/issues/3867)) + ## [16.2.7] - 2026-06-30 ### Breaking Changes diff --git a/packages/coding-agent/src/edit/hashline/filesystem.ts b/packages/coding-agent/src/edit/hashline/filesystem.ts index 63b9ffd17..765854275 100644 --- a/packages/coding-agent/src/edit/hashline/filesystem.ts +++ b/packages/coding-agent/src/edit/hashline/filesystem.ts @@ -123,6 +123,16 @@ export class HashlineFilesystem extends Filesystem { return content; } + async readBinary(relativePath: string): Promise { + const absolutePath = this.resolveAbsolute(relativePath); + try { + return await fs.readFile(absolutePath); + } catch (error) { + if (isEnoent(error)) throw new NotFoundError(relativePath, error); + throw error; + } + } + async preflightWrite(relativePath: string, options?: PreflightWriteOptions): Promise { const fileOp = options?.fileOp; if (fileOp?.kind === "rem") { diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index fad0130ae..1a09e9a50 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -111,6 +111,23 @@ describe("hashline executor", () => { }); }); + it("preserves UTF-8 BOM bytes when hashline edits decoded text", async () => { + await withTempDir(async tempDir => { + const filePath = path.join(tempDir, "Program.cs"); + const source = "using A;\n"; + await Bun.write(filePath, new Uint8Array([0xef, 0xbb, 0xbf, ...new TextEncoder().encode(source)])); + const session = makeHashlineSession(tempDir); + const sourceTag = recordFullSnapshot(getFileReadCache(session), filePath, source); + const input = `${header("Program.cs", sourceTag)}\n${sameLineRange(tag(1, source))}\n${repl("using B;")}\n`; + + await executeHashlineSingle(hashlineExecuteOptions(tempDir, input, undefined, session)); + + const bytes = await fs.readFile(filePath); + expect(Array.from(bytes.subarray(0, 3))).toEqual([0xef, 0xbb, 0xbf]); + expect(new TextDecoder().decode(bytes.subarray(3))).toBe("using B;\n"); + }); + }); + it("emits an actionable no-op diagnostic when the payload matches the file byte-for-byte", async () => { await withTempDir(async tempDir => { const filePath = path.join(tempDir, "a.ts"); diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index bc5589f3e..f2d477c34 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed hashline writes preserving UTF-8 BOM bytes when the host text decoder hides the leading `U+FEFF`. ([#3867](https://github.com/can1357/oh-my-pi/issues/3867)) + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/hashline/src/fs.ts b/packages/hashline/src/fs.ts index c1ec70ad0..425e78576 100644 --- a/packages/hashline/src/fs.ts +++ b/packages/hashline/src/fs.ts @@ -65,6 +65,9 @@ export abstract class Filesystem { /** Read the file's full text content. Throw on missing file. */ abstract readText(path: string): Promise; + /** Read the file's raw bytes when text decoding may hide leading bytes such as a UTF-8 BOM. */ + readBinary?(path: string): Promise; + /** Validate that `path` is writable before a prepared batch starts committing. */ async preflightWrite(_path: string, _options?: PreflightWriteOptions): Promise {} @@ -196,6 +199,15 @@ export class NodeFilesystem extends Filesystem { return file.text(); } + async readBinary(path: string): Promise { + try { + return await fs.readFile(path); + } catch (error) { + if (isNotFound(error)) throw new NotFoundError(path, error); + throw error; + } + } + async writeText(path: string, content: string): Promise { await Bun.write(path, content); return { text: content }; diff --git a/packages/hashline/src/patcher.ts b/packages/hashline/src/patcher.ts index ba5cc9a75..d87585756 100644 --- a/packages/hashline/src/patcher.ts +++ b/packages/hashline/src/patcher.ts @@ -148,6 +148,10 @@ function mergeWarnings(...sources: ReadonlyArray) return out; } +function hasUtf8Bom(bytes: Uint8Array | undefined): boolean { + return bytes !== undefined && bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf; +} + function assertUniqueCanonicalPaths(prepared: readonly PreparedSection[]): void { const seen = new Map(); for (const entry of prepared) { @@ -295,7 +299,8 @@ export class Patcher { throw new Error(`MV destination is the same as ${target.path}.`); } - const { bom, text } = stripBom(read.rawContent); + const { bom: bomFromText, text } = stripBom(read.rawContent); + const bom = bomFromText || (await this.#readBinaryBom(target.path)); const lineEnding = detectLineEnding(text); const normalized = normalizeToLF(text); @@ -453,6 +458,12 @@ export class Patcher { }; } + async #readBinaryBom(path: string): Promise { + if (!this.fs.readBinary) return ""; + const bytes = await this.fs.readBinary(path); + return hasUtf8Bom(bytes) ? "\uFEFF" : ""; + } + async #tryRead(path: string): Promise<{ exists: boolean; rawContent: string }> { try { const content = await this.fs.readText(path); diff --git a/packages/hashline/test/patcher.test.ts b/packages/hashline/test/patcher.test.ts index e4758025e..a9ccf75ed 100644 --- a/packages/hashline/test/patcher.test.ts +++ b/packages/hashline/test/patcher.test.ts @@ -1,10 +1,15 @@ import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { computeFileHash, + formatHashlineHeader, HEADTAIL_DRIFT_WARNING, InMemoryFilesystem, InMemorySnapshotStore, MismatchError, + NodeFilesystem, Patch, Patcher, } from "@oh-my-pi/hashline"; @@ -33,6 +38,26 @@ describe("Patcher snapshot tag integrity", () => { expect(fs.get(PATH)).toBe("after\n"); }); + it("restores a UTF-8 BOM hidden by Bun text decoding", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "hashline-bom-")); + try { + const filePath = path.join(tempDir, "Program.cs"); + const source = "using A;\n"; + await Bun.write(filePath, new Uint8Array([0xef, 0xbb, 0xbf, ...new TextEncoder().encode(source)])); + const snapshots = new InMemorySnapshotStore(); + const tag = snapshots.record(filePath, source); + const patch = Patch.parse([formatHashlineHeader(filePath, tag), "SWAP 1.=1:", "+using B;"].join("\n")); + + await new Patcher({ fs: new NodeFilesystem(), snapshots }).apply(patch); + + const bytes = await fs.readFile(filePath); + expect(Array.from(bytes.subarray(0, 3))).toEqual([0xef, 0xbb, 0xbf]); + expect(new TextDecoder().decode(bytes.subarray(3))).toBe("using B;\n"); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("validates any anchor purely from the content hash, even with no recorded snapshot", async () => { // The core fix: the tag fingerprints the WHOLE file. An edit anchored at // a line the model never saw recorded applies whenever the live file From 45d2bf86417707f532f8e136b712887ee9e3d4ac Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:49:41 +0000 Subject: [PATCH 45/88] style: bun run fix --- packages/coding-agent/test/tools/web-search-duckduckgo.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts index 4da54bddd..8c3ea24be 100644 --- a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts +++ b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts @@ -73,7 +73,7 @@ describe("DuckDuckGo web search provider", () => { const headers = capturedInit?.headers as Record; expect(headers["Content-Type"]).toBe("application/x-www-form-urlencoded"); expect(headers["User-Agent"]).toContain("Mozilla/5.0"); - expect(headers["Referer"]).toBe("https://html.duckduckgo.com/"); + expect(headers.Referer).toBe("https://html.duckduckgo.com/"); expect(headers["Accept-Language"]).toContain("en"); expect(headers["Sec-Fetch-Mode"]).toBe("navigate"); expect(headers["Sec-Ch-Ua"]).toContain("Chromium"); From 39688620f82806c1ac34cc422118567f819b936a Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:50:12 +0000 Subject: [PATCH 46/88] fix(compaction): kept newest snapcompact frames When legacy snapcompact archives exceed the per-request byte budget, retain frames from the newest end of the archived middle and restore oldest-to-newest order for the kept subset. --- .../test/session-manager/build-context.test.ts | 13 +++++++++---- packages/snapcompact/src/snapcompact.ts | 11 +++++++---- 2 files changed, 16 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/test/session-manager/build-context.test.ts b/packages/coding-agent/test/session-manager/build-context.test.ts index f703f4138..655040dbf 100644 --- a/packages/coding-agent/test/session-manager/build-context.test.ts +++ b/packages/coding-agent/test/session-manager/build-context.test.ts @@ -232,14 +232,15 @@ describe("buildSessionContext", () => { }); it("caps snapcompact frame payload in LLM context but preserves transcript frames", () => { - const largeFrame = "a".repeat(Math.ceil(snapcompact.FRAME_DATA_BYTES_BUDGET / 2) + 1); + const oldFrame = "o".repeat(Math.ceil(snapcompact.FRAME_DATA_BYTES_BUDGET / 2) + 1); + const newFrame = "n".repeat(oldFrame.length); const compacted: CompactionEntry = { ...compaction("3", "2", "Snapcompact summary", "1"), preserveData: { [snapcompact.PRESERVE_KEY]: { frames: [ - { data: largeFrame, mimeType: "image/png", cols: 10, rows: 10, chars: 10 }, - { data: largeFrame, mimeType: "image/png", cols: 10, rows: 10, chars: 10 }, + { data: oldFrame, mimeType: "image/png", cols: 10, rows: 10, chars: 10 }, + { data: newFrame, mimeType: "image/png", cols: 10, rows: 10, chars: 10 }, ], totalChars: 20, truncatedChars: 0, @@ -258,7 +259,11 @@ describe("buildSessionContext", () => { const llmContext = buildSessionContext(entries); const summary = llmContext.messages[0]; if (summary?.role !== "compactionSummary") throw new Error("Expected LLM compaction summary"); - expect(summary.blocks?.filter(block => block.type === "image")).toHaveLength(1); + const imageBlocks = summary.blocks?.filter(block => block.type === "image"); + expect(imageBlocks).toHaveLength(1); + const keptImage = imageBlocks?.[0]; + if (keptImage?.type !== "image") throw new Error("Expected kept snapcompact image"); + expect(keptImage.data).toBe(newFrame); expect( summary.blocks?.some(block => block.type === "text" && block.text.includes("image middle omitted")), ).toBe(true); diff --git a/packages/snapcompact/src/snapcompact.ts b/packages/snapcompact/src/snapcompact.ts index b95f71e46..d05586950 100644 --- a/packages/snapcompact/src/snapcompact.ts +++ b/packages/snapcompact/src/snapcompact.ts @@ -1359,8 +1359,10 @@ function imagesWithinBudget( let usedBytes = 0; let omittedFrames = 0; let omittedBytes = 0; - const kept: Frame[] = []; - for (const frame of archive.frames) { + const keptNewestFirst: Frame[] = []; + for (let index = archive.frames.length - 1; index >= 0; index--) { + const frame = archive.frames[index]; + if (!frame) continue; const bytes = frame.data.length; if (usedBytes + bytes > maxFrameDataBytes) { omittedFrames++; @@ -1368,9 +1370,10 @@ function imagesWithinBudget( continue; } usedBytes += bytes; - kept.push(frame); + keptNewestFirst.push(frame); } - return { images: images({ ...archive, frames: kept }), omittedFrames, omittedBytes }; + keptNewestFirst.reverse(); + return { images: images({ ...archive, frames: keptNewestFirst }), omittedFrames, omittedBytes }; } function omittedFrameNotice(omittedFrames: number, omittedBytes: number): string { From d1e412eeff41f334630da6e2eca00015b6d93ff6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:51:53 +0000 Subject: [PATCH 47/88] Revert "style: bun run fix" This reverts commit 45d2bf86417707f532f8e136b712887ee9e3d4ac. --- packages/coding-agent/test/tools/web-search-duckduckgo.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts index 8c3ea24be..4da54bddd 100644 --- a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts +++ b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts @@ -73,7 +73,7 @@ describe("DuckDuckGo web search provider", () => { const headers = capturedInit?.headers as Record; expect(headers["Content-Type"]).toBe("application/x-www-form-urlencoded"); expect(headers["User-Agent"]).toContain("Mozilla/5.0"); - expect(headers.Referer).toBe("https://html.duckduckgo.com/"); + expect(headers["Referer"]).toBe("https://html.duckduckgo.com/"); expect(headers["Accept-Language"]).toContain("en"); expect(headers["Sec-Fetch-Mode"]).toBe("navigate"); expect(headers["Sec-Ch-Ua"]).toContain("Chromium"); From dd4eb68b378658a3eea2fca08b2485093c4464ed Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:52:02 +0000 Subject: [PATCH 48/88] style: bun run fix --- packages/coding-agent/test/tools/web-search-duckduckgo.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts index 4da54bddd..8c3ea24be 100644 --- a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts +++ b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts @@ -73,7 +73,7 @@ describe("DuckDuckGo web search provider", () => { const headers = capturedInit?.headers as Record; expect(headers["Content-Type"]).toBe("application/x-www-form-urlencoded"); expect(headers["User-Agent"]).toContain("Mozilla/5.0"); - expect(headers["Referer"]).toBe("https://html.duckduckgo.com/"); + expect(headers.Referer).toBe("https://html.duckduckgo.com/"); expect(headers["Accept-Language"]).toContain("en"); expect(headers["Sec-Fetch-Mode"]).toBe("navigate"); expect(headers["Sec-Ch-Ua"]).toContain("Chromium"); From 7ec00a50076a39e84174969847cea813d3968875 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 05:58:01 +0000 Subject: [PATCH 49/88] fix(edit): skipped notebook bom sniffing Skipped binary BOM detection for notebook-backed hashline reads so virtual cell text is serialized without a leading U+FEFF marker. --- .../src/edit/hashline/filesystem.ts | 4 ++- .../coding-agent/test/core/hashline.test.ts | 33 +++++++++++++++++++ packages/hashline/src/fs.ts | 4 +-- 3 files changed, 38 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/edit/hashline/filesystem.ts b/packages/coding-agent/src/edit/hashline/filesystem.ts index 765854275..41f15842b 100644 --- a/packages/coding-agent/src/edit/hashline/filesystem.ts +++ b/packages/coding-agent/src/edit/hashline/filesystem.ts @@ -28,6 +28,7 @@ import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation"; import { isInternalUrlPath } from "../../tools/path-utils"; import { enforcePlanModeWrite, resolvePlanPath, targetsLocalSandbox } from "../../tools/plan-mode-guard"; import { canonicalSnapshotKey } from "../file-snapshot-store"; +import { isNotebookPath } from "../notebook"; import { readEditFileText, serializeEditFileText } from "../read-file"; import type { LspBatchRequest } from "../renderer"; @@ -123,8 +124,9 @@ export class HashlineFilesystem extends Filesystem { return content; } - async readBinary(relativePath: string): Promise { + async readBinary(relativePath: string): Promise { const absolutePath = this.resolveAbsolute(relativePath); + if (isNotebookPath(absolutePath)) return undefined; try { return await fs.readFile(absolutePath); } catch (error) { diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 1a09e9a50..6627351ff 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -128,6 +128,39 @@ describe("hashline executor", () => { }); }); + it("edits BOM-prefixed notebooks through the virtual cell text", async () => { + await withTempDir(async tempDir => { + const filePath = path.join(tempDir, "notebook.ipynb"); + const notebook = { + cells: [ + { + cell_type: "markdown", + metadata: { keep: true }, + source: ["# Title\n"], + }, + ], + metadata: {}, + nbformat: 4, + nbformat_minor: 5, + }; + await Bun.write( + filePath, + new Uint8Array([0xef, 0xbb, 0xbf, ...new TextEncoder().encode(JSON.stringify(notebook))]), + ); + const session = makeHashlineSession(tempDir); + const editableText = "# %% [markdown] cell:0\n# Title\n"; + const sourceTag = recordFullSnapshot(getFileReadCache(session), filePath, editableText); + const input = `${header("notebook.ipynb", sourceTag)}\n${sameLineRange(tag(2, "# Title"))}\n${repl("# Updated")}\n`; + + await executeHashlineSingle(hashlineExecuteOptions(tempDir, input, undefined, session)); + + const updated = await Bun.file(filePath).json(); + expect(updated.cells).toHaveLength(1); + expect(updated.cells[0].source).toEqual(["# Updated\n"]); + expect(updated.cells[0].metadata).toEqual({ keep: true }); + }); + }); + it("emits an actionable no-op diagnostic when the payload matches the file byte-for-byte", async () => { await withTempDir(async tempDir => { const filePath = path.join(tempDir, "a.ts"); diff --git a/packages/hashline/src/fs.ts b/packages/hashline/src/fs.ts index 425e78576..bca93496a 100644 --- a/packages/hashline/src/fs.ts +++ b/packages/hashline/src/fs.ts @@ -65,8 +65,8 @@ export abstract class Filesystem { /** Read the file's full text content. Throw on missing file. */ abstract readText(path: string): Promise; - /** Read the file's raw bytes when text decoding may hide leading bytes such as a UTF-8 BOM. */ - readBinary?(path: string): Promise; + /** Read raw bytes for backends whose text is a direct decode of persisted bytes. */ + readBinary?(path: string): Promise; /** Validate that `path` is writable before a prepared batch starts committing. */ async preflightWrite(_path: string, _options?: PreflightWriteOptions): Promise {} From 4810f47db9cbb6901d67ea68914e9ce9453857dc Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 30 Jun 2026 06:20:03 +0000 Subject: [PATCH 50/88] fix(yield): validate incremental sections per-label to prevent fatal schema_violation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The yield tool's per-call schema validator was skipped entirely for incremental yields (`type: ["