From 4a6362192dc3e325f486101f033c2342003877b0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 06:33:53 +0000 Subject: [PATCH 01/29] fix(legacy-pi-compat): validated package-root override targets before rewrite MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The package-root override branch of `resolveCanonicalPiSpecifier` returned the bunfs override path without checking the target was actually present, so when `bun --compile` quietly dropped one of the extra entrypoints (observed on macOS arm64 release binaries), the static rewrite emitted a `file://` URL to a missing module. The #1216 fallback only fired on the throwing `getResolvedSpecifier` path, so the override never threw and the rewrite committed the bad URL — extensions silently failed to load. Each override target is now checked with `fs.existsSync` at module init. Missing entries are dropped from `LEGACY_PI_PACKAGE_ROOT_OVERRIDES` so `resolveCanonicalPiSpecifier` falls through to `getResolvedSpecifier`, which throws under bunfs and triggers the existing rewrite catch — Bun then resolves the canonical `@oh-my-pi/pi-*` specifier from the extension's own `node_modules`. Fixes #2168 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../extensibility/plugins/legacy-pi-compat.ts | 36 ++++++++++--- .../legacy-pi-override-fallback.test.ts | 52 +++++++++++++++++++ 3 files changed, 86 insertions(+), 6 deletions(-) create mode 100644 packages/coding-agent/test/extensibility/legacy-pi-override-fallback.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bef9ae4ac..fe19290c3 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed compiled-binary extensions failing to load `@oh-my-pi/pi-*` packages when `bun --compile` quietly dropped one of the extra entrypoints (observed on macOS arm64 release builds): the legacy-pi compat shim's package-root override branch returned the bunfs path without checking the target was present, so the rewrite emitted a `file://` URL to a missing module and the #1216 fallback (scoped to the throwing `getResolvedSpecifier` path) never ran. Override targets are now validated against the on-disk filesystem at module init, missing entries are dropped, and resolution falls through to canonical lookup so Bun resolves the import from the extension's own `node_modules` ([#2168](https://github.com/can1357/oh-my-pi/issues/2168)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index e85e63c7d..31c6da921 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -1,4 +1,4 @@ -import * as fs from "node:fs/promises"; +import * as fs from "node:fs"; import * as path from "node:path"; import * as url from "node:url"; import { isCompiledBinary } from "@oh-my-pi/pi-utils"; @@ -142,7 +142,31 @@ const LEGACY_PI_CODING_AGENT_SHIM_PATH = BUNFS_PACKAGE_ROOT // `Bun.resolveSync`, and hardcoding a relative source-tree path would break // installs where the bundled packages live at `node_modules/@oh-my-pi/pi-*` // rather than `packages/*`. -const LEGACY_PI_PACKAGE_ROOT_OVERRIDES: Record = { +// +// Every override target is validated against the on-disk filesystem at module +// init: any entry whose file is missing (e.g. a compiled binary where Bun's +// `--compile` quietly dropped an additional entrypoint — issue #2168) is left +// out so `resolveCanonicalPiSpecifier` falls through to `getResolvedSpecifier`, +// which throws under bunfs and triggers the catch in `rewriteLegacyPiImports`. +// That catch leaves the specifier untouched so Bun resolves the canonical +// `@oh-my-pi/pi-*` import from the extension's own `node_modules` instead of +// emitting a bunfs `file://` URL to a module that isn't actually present. + +/** + * Drop overrides whose targets are missing on disk so they can fall through to + * the canonical-resolution path. Exported for the test seam in #2168. + * + * `pathExistsSync` defaults to `fs.existsSync`; the tests inject a stub to + * simulate the missing-entrypoint failure mode without touching the real FS. + */ +export function __validateLegacyPiPackageRootOverrides( + candidates: Record, + pathExistsSync: (p: string) => boolean = fs.existsSync, +): Record { + return Object.fromEntries(Object.entries(candidates).filter(([, candidate]) => pathExistsSync(candidate))); +} + +const LEGACY_PI_PACKAGE_ROOT_OVERRIDES = __validateLegacyPiPackageRootOverrides({ [`${CANONICAL_PI_SCOPE}/pi-ai`]: LEGACY_PI_AI_SHIM_PATH, [`${CANONICAL_PI_SCOPE}/pi-coding-agent`]: LEGACY_PI_CODING_AGENT_SHIM_PATH, ...(BUNFS_PACKAGE_ROOT @@ -153,7 +177,7 @@ const LEGACY_PI_PACKAGE_ROOT_OVERRIDES: Record = { [`${CANONICAL_PI_SCOPE}/pi-utils`]: bunfsPath("utils", "src", "index.js"), } : {}), -}; +}); let isLegacyPiSpecifierShimInstalled = false; @@ -253,7 +277,7 @@ function isRecord(value: unknown): value is Record { async function pathExists(p: string): Promise { try { - await fs.stat(p); + await fs.promises.stat(p); return true; } catch { return false; @@ -267,7 +291,7 @@ function hasSourceModuleExtension(p: string): boolean { async function resolveSourceModuleFile(basePath: string): Promise { try { - const stats = await fs.stat(basePath); + const stats = await fs.promises.stat(basePath); if (stats.isFile()) { // Non-source files (JSON, WASM, text assets, etc.) bypass the on-load // rewrite hook so Bun's native loaders handle them; our hook would @@ -475,7 +499,7 @@ const hookedExtensionEntries = new Set(); /** Resolve symlinks in a path, falling back to the input if realpath fails. */ async function realpathOrSelf(p: string): Promise { try { - return await fs.realpath(p); + return await fs.promises.realpath(p); } catch { return p; } diff --git a/packages/coding-agent/test/extensibility/legacy-pi-override-fallback.test.ts b/packages/coding-agent/test/extensibility/legacy-pi-override-fallback.test.ts new file mode 100644 index 000000000..372ed6e25 --- /dev/null +++ b/packages/coding-agent/test/extensibility/legacy-pi-override-fallback.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "bun:test"; +import { __validateLegacyPiPackageRootOverrides } from "@oh-my-pi/pi-coding-agent/extensibility/plugins/legacy-pi-compat"; + +// Regression for issue #2168: in compiled-binary mode the package-root +// override branch of `resolveCanonicalPiSpecifier` returned a bunfs path +// without checking the target was actually present. When `bun --compile` +// quietly dropped one of the extra entrypoints (observed on macOS arm64 +// release builds), the rewrite still emitted a `file://` URL to a missing +// module, defeating the #1216 fallback that only fired on the throwing +// `getResolvedSpecifier` path. The fix validates each override at module +// init so missing entries fall through to canonical resolution and Bun +// resolves the import from the extension's own `node_modules`. +describe("legacy pi compat package-root override validation (issue #2168)", () => { + it("keeps overrides whose targets exist", () => { + const candidates = { + "@oh-my-pi/pi-ai": "/tmp/exists-ai.js", + "@oh-my-pi/pi-utils": "/tmp/exists-utils.js", + }; + const result = __validateLegacyPiPackageRootOverrides(candidates, () => true); + expect(result).toEqual(candidates); + }); + + it("drops overrides whose targets are missing on disk", () => { + const candidates = { + "@oh-my-pi/pi-ai": "/tmp/exists-ai.js", + "@oh-my-pi/pi-coding-agent": "/tmp/exists-shim.js", + "@oh-my-pi/pi-utils": "/$bunfs/root/packages/utils/src/index.js", + "@oh-my-pi/pi-tui": "/$bunfs/root/packages/tui/src/index.js", + }; + const missing = new Set(["/$bunfs/root/packages/utils/src/index.js", "/$bunfs/root/packages/tui/src/index.js"]); + const result = __validateLegacyPiPackageRootOverrides(candidates, p => !missing.has(p)); + expect(result).toEqual({ + "@oh-my-pi/pi-ai": "/tmp/exists-ai.js", + "@oh-my-pi/pi-coding-agent": "/tmp/exists-shim.js", + }); + // `pi-utils` and `pi-tui` are absent so the resolver falls through to + // `getResolvedSpecifier` (which throws under bunfs), which triggers + // the catch in `rewriteLegacyPiImports` that leaves the specifier + // unchanged for native `node_modules` resolution. + expect(result).not.toHaveProperty("@oh-my-pi/pi-utils"); + expect(result).not.toHaveProperty("@oh-my-pi/pi-tui"); + }); + + it("drops every override when none of the targets exist", () => { + const candidates = { + "@oh-my-pi/pi-utils": "/$bunfs/root/packages/utils/src/index.js", + "@oh-my-pi/pi-tui": "/$bunfs/root/packages/tui/src/index.js", + }; + const result = __validateLegacyPiPackageRootOverrides(candidates, () => false); + expect(result).toEqual({}); + }); +}); From 906742bc293046da462686d058625b00c4296f6c Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 06:38:49 +0000 Subject: [PATCH 02/29] fix(tui): wrap long slash-command picker descriptions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The slash-command autocomplete picker rendered each command/skill description on a single line and clipped it. Other long TUI text (markdown, settings-list description pane, the Text component) already wraps via wrapTextWithAnsi — the picker did not, so descriptions like the bundled improve-codebase-architecture skill were unreadable at normal terminal widths. SelectList now exposes an opt-in wrapDescription layout flag. When set, #renderItem returns one entry per visual row, with continuation rows indented to descriptionStart so the wrapped text stays column-aligned under the description column. render() flattens the per-item arrays into the ScrollView buffer and tracks visual rows for the scrollbar thumb so it stays accurate when items wrap unevenly. The narrow-width fallback (width <= 40) is preserved and other SelectList consumers (model picker, settings list, ask_user pickers) keep the existing single-line truncation behavior because the flag is off by default. editor.ts enables it on SLASH_COMMAND_SELECT_LIST_LAYOUT. Tests cover short-description single-row, long-description wrap with every row within the picker width, continuation-row indent alignment, the narrow-width fallback, arrow-down advancing exactly one item even when the prior item wrapped, and scrollbar visibility under wrap. Fixes #2169 --- packages/tui/CHANGELOG.md | 4 + packages/tui/src/components/editor.ts | 1 + packages/tui/src/components/select-list.ts | 139 +++++++++++++++++---- packages/tui/test/select-list.test.ts | 82 ++++++++++++ 4 files changed, 202 insertions(+), 24 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index e90de40d2..457bd1ad6 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added a `wrapDescription` option to `SelectListLayoutOptions`. When enabled, long descriptions wrap onto continuation rows indented under the description column instead of being silently truncated. The slash-command/skill autocomplete picker now opts in so descriptions like the bundled skills' remain fully readable at normal terminal widths. Navigation stays item-to-item, the narrow-width fallback (`width <= 40`) is unchanged, and the `ScrollView` scrollbar tracks visual rows so the thumb stays correct when items wrap unevenly. ([#2169](https://github.com/can1357/oh-my-pi/issues/2169)) + ## [15.10.8] - 2026-06-09 ### Fixed diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 7725f2a42..8dbb89d18 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -27,6 +27,7 @@ const SLASH_COMMAND_SELECT_LIST_LAYOUT: SelectListLayoutOptions = { minPrimaryColumnWidth: 12, maxPrimaryColumnWidth: 32, overflowSearch: false, + wrapDescription: true, }; function sanitizeLoadedText(text: string): string { diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index dfe95a920..e17211327 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -3,7 +3,7 @@ import { getKeybindings } from "../keybindings"; import { extractPrintableText } from "../keys"; import type { SymbolTheme } from "../symbols"; import type { Component } from "../tui"; -import { Ellipsis, padding, replaceTabs, truncateToWidth, visibleWidth } from "../utils"; +import { Ellipsis, padding, replaceTabs, truncateToWidth, visibleWidth, wrapTextWithAnsi } from "../utils"; import { ScrollView } from "./scroll-view"; const DEFAULT_PRIMARY_COLUMN_WIDTH = 32; @@ -50,8 +50,33 @@ export interface SelectListLayoutOptions { truncatePrimary?: (context: SelectListTruncatePrimaryContext) => string; /** Enable type-to-filter search when the item count exceeds maxVisible. Defaults to true. */ overflowSearch?: boolean; + /** + * Wrap long descriptions onto continuation rows indented under the + * description column instead of truncating. Defaults to false so existing + * single-line consumers are unaffected. Navigation remains item-to-item; + * the scrollbar tracks visual rows so the thumb stays correct when items + * wrap unevenly. + */ + wrapDescription?: boolean; } +type SelectItemLayout = + | { + kind: "description"; + prefix: string; + truncatedValue: string; + spacing: string; + descriptionSingleLine: string; + descriptionStart: number; + remainingWidth: number; + } + | { + kind: "primary"; + prefix: string; + truncatedValue: string; + spacing: ""; + }; + export class SelectList implements Component { #filteredItems: ReadonlyArray; #filterQuery = ""; @@ -107,23 +132,34 @@ export class SelectList implements Component { // Render visible items const overflow = this.#filteredItems.length > this.maxVisible; const rowWidth = Math.max(0, width - (overflow ? 1 : 0)); + const wrapEnabled = this.layout.wrapDescription === true; const rows: string[] = []; - for (let i = startIndex; i < endIndex; i++) { + let visualOffset = 0; + let visualTotal = 0; + for (let i = 0; i < this.#filteredItems.length; i++) { const item = this.#filteredItems[i]; if (!item) continue; - - const isSelected = i === this.#selectedIndex; - const descriptionText = item.description ? sanitizeSingleLine(item.description) : undefined; - rows.push(this.#renderItem(item, isSelected, rowWidth, descriptionText, primaryColumnWidth)); + let rowCount = 1; + if (i >= startIndex && i < endIndex) { + const itemRows = this.#renderItem(item, i === this.#selectedIndex, rowWidth, primaryColumnWidth); + rowCount = itemRows.length; + rows.push(...itemRows); + } else if (wrapEnabled) { + // Wrap is opt-in, so non-wrap layouts keep the original + // constant-time path for items outside the visible window. + rowCount = this.#computeItemRowCount(item, rowWidth, primaryColumnWidth); + } + if (i < startIndex) visualOffset += rowCount; + visualTotal += rowCount; } const sv = new ScrollView(rows, { height: rows.length, scrollbar: "auto", - totalRows: this.#filteredItems.length, + totalRows: visualTotal, theme: { track: t => this.theme.scrollInfo(t), thumb: t => this.theme.selectedPrefix(t) }, }); - sv.setScrollOffset(startIndex); + sv.setScrollOffset(visualOffset); lines.push(...sv.render(width)); // Add search status when relevant (scrollbar now indicates overflow) @@ -182,13 +218,65 @@ export class SelectList implements Component { item: SelectItem, isSelected: boolean, width: number, - descriptionSingleLine: string | undefined, primaryColumnWidth: number, - ): string { + ): string[] { + const layout = this.#computeItemLayout(item, isSelected, width, primaryColumnWidth); + const { prefix, truncatedValue, spacing } = layout; + + if (layout.kind === "description") { + const { descriptionSingleLine, descriptionStart, remainingWidth } = layout; + if (this.layout.wrapDescription) { + const wrapped = wrapTextWithAnsi(descriptionSingleLine, remainingWidth); + if (wrapped.length === 0) wrapped.push(""); + const indent = padding(descriptionStart); + const first = wrapped[0] ?? ""; + if (isSelected) { + const rows = [this.theme.selectedText(`${prefix}${truncatedValue}${spacing}${first}`)]; + for (let i = 1; i < wrapped.length; i++) { + rows.push(this.theme.selectedText(`${indent}${wrapped[i]}`)); + } + return rows; + } + const rows = [prefix + truncatedValue + this.theme.description(spacing + first)]; + for (let i = 1; i < wrapped.length; i++) { + rows.push(this.theme.description(`${indent}${wrapped[i]}`)); + } + return rows; + } + + const truncatedDesc = truncateToWidth(descriptionSingleLine, remainingWidth, Ellipsis.Omit); + if (isSelected) { + return [this.theme.selectedText(`${prefix}${truncatedValue}${spacing}${truncatedDesc}`)]; + } + return [prefix + truncatedValue + this.theme.description(spacing + truncatedDesc)]; + } + + if (isSelected) { + return [this.theme.selectedText(`${prefix}${truncatedValue}`)]; + } + return [prefix + truncatedValue]; + } + + #computeItemRowCount(item: SelectItem, width: number, primaryColumnWidth: number): number { + // Selection style does not change row count; pass isSelected=false to + // keep the cheap path uniform for items outside the visible window. + const layout = this.#computeItemLayout(item, false, width, primaryColumnWidth); + if (layout.kind !== "description") return 1; + const wrapped = wrapTextWithAnsi(layout.descriptionSingleLine, layout.remainingWidth); + return Math.max(1, wrapped.length); + } + + #computeItemLayout( + item: SelectItem, + isSelected: boolean, + width: number, + primaryColumnWidth: number, + ): SelectItemLayout { const prefix = isSelected ? `${this.theme.symbols.cursor} ` : padding(visibleWidth(this.theme.symbols.cursor) + 1); const prefixWidth = visibleWidth(prefix); + const descriptionSingleLine = item.description ? sanitizeSingleLine(item.description) : undefined; if (descriptionSingleLine && width > 40) { const effectivePrimaryColumnWidth = Math.max(1, Math.min(primaryColumnWidth, width - prefixWidth - 4)); @@ -200,23 +288,26 @@ export class SelectList implements Component { const remainingWidth = width - descriptionStart - 2; // -2 for safety if (remainingWidth > MIN_DESCRIPTION_WIDTH) { - const truncatedDesc = truncateToWidth(descriptionSingleLine, remainingWidth, Ellipsis.Omit); - if (isSelected) { - return this.theme.selectedText(`${prefix}${truncatedValue}${spacing}${truncatedDesc}`); - } - - const descText = this.theme.description(spacing + truncatedDesc); - return prefix + truncatedValue + descText; + return { + kind: "description", + prefix, + truncatedValue, + spacing, + descriptionSingleLine, + descriptionStart, + remainingWidth, + }; } } - const maxWidth = width - prefixWidth - 2; - const truncatedValue = this.#truncatePrimary(item, isSelected, maxWidth, maxWidth); - if (isSelected) { - return this.theme.selectedText(`${prefix}${truncatedValue}`); - } - - return prefix + truncatedValue; + const fallbackMax = width - prefixWidth - 2; + const truncatedValue = this.#truncatePrimary(item, isSelected, fallbackMax, fallbackMax); + return { + kind: "primary", + prefix, + truncatedValue, + spacing: "", + }; } #getPrimaryColumnWidth(): number { diff --git a/packages/tui/test/select-list.test.ts b/packages/tui/test/select-list.test.ts index 098584b01..684b6ba98 100644 --- a/packages/tui/test/select-list.test.ts +++ b/packages/tui/test/select-list.test.ts @@ -225,4 +225,86 @@ describe("SelectList", () => { expect(list.render(40).join("\n")).not.toContain("█"); }); + + describe("wrapDescription", () => { + const longDescription = + "Plan and execute non-trivial architectural improvements to the codebase. Use this skill when you need to refactor existing systems, restructure modules, or change interfaces across multiple files."; + + it("keeps short descriptions on a single row", () => { + const items = [{ value: "short", label: "short", description: "fits easily" }]; + const list = new SelectList(items, 5, testTheme, { wrapDescription: true }); + + const rendered = list.render(80); + + expect(rendered).toHaveLength(1); + expect(rendered[0]).toContain("fits easily"); + }); + + it("wraps long descriptions under the description column", () => { + const items = [{ value: "long", label: "long-skill-name", description: longDescription }]; + const list = new SelectList(items, 5, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 32, + wrapDescription: true, + }); + + const rendered = list.render(80); + + // Long description must materialize as multiple rows; truncation would + // silently drop the tail (the issue). + expect(rendered.length).toBeGreaterThan(1); + // Every visual row must fit within the picker width. + for (const row of rendered) { + expect(visibleWidth(row)).toBeLessThanOrEqual(80); + } + // The first row carries the primary label and the wrapped tail must + // reach the closing words of the description. + expect(rendered[0]).toContain("long-skill-name"); + expect(rendered.join("\n")).toContain("across multiple files."); + // Continuation rows align under the description column. The cursor + // column on the first row is occupied; continuation rows lead with + // spaces up to the same offset where the description starts. + const descStart = visibleIndexOf(rendered[0], "Plan"); + for (let i = 1; i < rendered.length; i++) { + expect(rendered[i].slice(0, descStart)).toBe(" ".repeat(descStart)); + } + }); + + it("falls back to the no-description layout at narrow widths", () => { + const items = [{ value: "long", label: "long", description: longDescription }]; + const list = new SelectList(items, 5, testTheme, { wrapDescription: true }); + + // width <= 40 trips the existing primary-only fallback. + const rendered = list.render(40); + + expect(rendered).toHaveLength(1); + expect(rendered[0]).not.toContain("Plan and execute"); + }); + + it("advances selection by one item even when items wrap", () => { + const items = [ + { value: "a", label: "a", description: longDescription }, + { value: "b", label: "b", description: "short" }, + ]; + const list = new SelectList(items, 5, testTheme, { wrapDescription: true }); + + expect(list.getSelectedItem()?.value).toBe("a"); + // Press Down once → second item, regardless of how many visual rows + // the first item spans. + list.handleInput("\x1b[B"); + expect(list.getSelectedItem()?.value).toBe("b"); + }); + + it("renders the scrollbar when wrapped items overflow the visible window", () => { + const items = Array.from({ length: 6 }, (_, i) => ({ + value: `v${i}`, + label: `Item ${i}`, + description: longDescription, + })); + const list = new SelectList(items, 3, testTheme, { wrapDescription: true }); + + const rendered = list.render(80).join("\n"); + expect(rendered).toContain("█"); + }); + }); }); From 16e96602af9e4503c2a06e299cab746166c31ab5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 06:39:10 +0000 Subject: [PATCH 03/29] style: bun run fix --- packages/tui/src/components/select-list.ts | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index e17211327..2988f44a2 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -214,12 +214,7 @@ export class SelectList implements Component { } } - #renderItem( - item: SelectItem, - isSelected: boolean, - width: number, - primaryColumnWidth: number, - ): string[] { + #renderItem(item: SelectItem, isSelected: boolean, width: number, primaryColumnWidth: number): string[] { const layout = this.#computeItemLayout(item, isSelected, width, primaryColumnWidth); const { prefix, truncatedValue, spacing } = layout; From 65b72b50b3480529bda9a18b17277d69a2ba3577 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 06:47:54 +0000 Subject: [PATCH 04/29] fix(tui): cap wrapped SelectList popup at maxVisible rows MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reviewer flagged that when wrapDescription was enabled, rows.length — fed as ScrollView height — could vastly exceed maxVisible (three 3-line wrapped skills became a 9-row popup), and the scrollbar stayed hidden whenever the filtered item count was <= maxVisible even though visual rows overflowed. maxVisible now denotes the picker's visual row budget. SelectList computes a per-item visual row count once, picks a window centered on the selected item that fits within the budget via a new #pickWindow helper, and caps the materialized rows array at the budget so a single oversize wrapped item still keeps the popup bounded — the scrollbar carries the offscreen tail. Non-wrap layouts thread through the same helper with every count fixed at 1, resolving to the same [selectedIndex - floor(maxVisible/2) .. + maxVisible) window the prior arithmetic produced. Tests added: popup capped at maxVisible across multiple wrapped items with scrollbar present, selected item stays visible when navigation shifts the window past the budget, and a single oversize wrapped item is clipped to maxVisible rows. --- packages/tui/CHANGELOG.md | 2 +- packages/tui/src/components/select-list.ts | 113 ++++++++++++++++----- packages/tui/test/select-list.test.ts | 65 ++++++++++++ 3 files changed, 154 insertions(+), 26 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 457bd1ad6..d05ad2d29 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added a `wrapDescription` option to `SelectListLayoutOptions`. When enabled, long descriptions wrap onto continuation rows indented under the description column instead of being silently truncated. The slash-command/skill autocomplete picker now opts in so descriptions like the bundled skills' remain fully readable at normal terminal widths. Navigation stays item-to-item, the narrow-width fallback (`width <= 40`) is unchanged, and the `ScrollView` scrollbar tracks visual rows so the thumb stays correct when items wrap unevenly. ([#2169](https://github.com/can1357/oh-my-pi/issues/2169)) +- Added a `wrapDescription` option to `SelectListLayoutOptions`. When enabled, long descriptions wrap onto continuation rows indented under the description column instead of being silently truncated. The slash-command/skill autocomplete picker now opts in so descriptions like the bundled skills' remain fully readable at normal terminal widths. `maxVisible` becomes the picker's visual row budget so the popup height stays bounded even when items wrap (a single 5-row description with `maxVisible=3` clips with the scrollbar carrying the offscreen tail). Navigation stays item-to-item, the narrow-width fallback (`width <= 40`) is unchanged, and the `ScrollView` scrollbar tracks visual rows so the thumb stays correct when items wrap unevenly. ([#2169](https://github.com/can1357/oh-my-pi/issues/2169)) ## [15.10.8] - 2026-06-09 diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index 2988f44a2..30633618d 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -121,36 +121,51 @@ export class SelectList implements Component { } const primaryColumnWidth = this.#getPrimaryColumnWidth(); - - // Calculate visible range with scrolling - const startIndex = Math.max( - 0, - Math.min(this.#selectedIndex - Math.floor(this.maxVisible / 2), this.#filteredItems.length - this.maxVisible), - ); - const endIndex = Math.min(startIndex + this.maxVisible, this.#filteredItems.length); - - // Render visible items - const overflow = this.#filteredItems.length > this.maxVisible; - const rowWidth = Math.max(0, width - (overflow ? 1 : 0)); const wrapEnabled = this.layout.wrapDescription === true; - const rows: string[] = []; - let visualOffset = 0; + // `maxVisible` is the picker's visual row budget. For non-wrap layouts + // every item is one row, so the budget matches the original item count. + const visualBudget = this.maxVisible; + + // Compute per-item visual row counts at the conservative width (i.e. + // assume the scrollbar column might be reserved). For non-wrap layouts + // every count is 1, so visualTotal == #filteredItems and overflow falls + // back to the original `N > maxVisible` predicate exactly. + const conservativeRowWidth = Math.max(0, width - 1); + const rowCounts = new Array(this.#filteredItems.length); let visualTotal = 0; for (let i = 0; i < this.#filteredItems.length; i++) { const item = this.#filteredItems[i]; - if (!item) continue; - let rowCount = 1; - if (i >= startIndex && i < endIndex) { - const itemRows = this.#renderItem(item, i === this.#selectedIndex, rowWidth, primaryColumnWidth); - rowCount = itemRows.length; - rows.push(...itemRows); - } else if (wrapEnabled) { - // Wrap is opt-in, so non-wrap layouts keep the original - // constant-time path for items outside the visible window. - rowCount = this.#computeItemRowCount(item, rowWidth, primaryColumnWidth); + if (!item) { + rowCounts[i] = 0; + continue; + } + rowCounts[i] = wrapEnabled + ? this.#computeItemRowCount(item, conservativeRowWidth, primaryColumnWidth) + : 1; + visualTotal += rowCounts[i]; + } + + const overflow = visualTotal > visualBudget; + const rowWidth = Math.max(0, width - (overflow ? 1 : 0)); + + // Pick a window centered on the selected item that fits in visualBudget + // rows. Falls through to the original item-count window when every row + // count is 1. + const { startIndex, endIndex, visualOffset } = this.#pickWindow(rowCounts, visualBudget); + + // Render visible items. Cap rows at the budget so a single item that + // wraps to more than `visualBudget` rows (pathological — e.g. a 5-row + // description with maxVisible=3) still keeps the popup bounded; the + // scrollbar carries the offscreen rows. + const rows: string[] = []; + for (let i = startIndex; i < endIndex && rows.length < visualBudget; i++) { + const item = this.#filteredItems[i]; + if (!item) continue; + const itemRows = this.#renderItem(item, i === this.#selectedIndex, rowWidth, primaryColumnWidth); + for (const row of itemRows) { + if (rows.length >= visualBudget) break; + rows.push(row); } - if (i < startIndex) visualOffset += rowCount; - visualTotal += rowCount; } const sv = new ScrollView(rows, { @@ -261,6 +276,54 @@ export class SelectList implements Component { return Math.max(1, wrapped.length); } + /** + * Pick a contiguous window of items containing `selectedIndex` such that + * their visual rows fit within `budget`. Centers the selection roughly + * mid-window: first expands up by ⌊budget/2⌋ rows, then fills downward, + * then back upward with any remaining budget. For non-wrap layouts (every + * `rowCounts[i] === 1`) this resolves to the same `[start, start+maxVisible)` + * window the prior arithmetic produced. + */ + #pickWindow( + rowCounts: ReadonlyArray, + budget: number, + ): { startIndex: number; endIndex: number; visualOffset: number } { + const n = rowCounts.length; + const selected = Math.max(0, Math.min(this.#selectedIndex, n - 1)); + if (n === 0) return { startIndex: 0, endIndex: 0, visualOffset: 0 }; + + const half = Math.floor(budget / 2); + let lo = selected; + let rowsAboveSelected = 0; + // Step 1: expand upward up to `half` rows above the selection so it + // lands near the visual middle, matching the prior centering. + while (lo > 0 && rowsAboveSelected + (rowCounts[lo - 1] ?? 0) <= half) { + lo--; + rowsAboveSelected += rowCounts[lo] ?? 0; + } + + // Step 2: expand downward until the budget is filled. The selected + // item's own rows are always counted; if it alone exceeds `budget` + // the surplus is clipped at render time and the scrollbar carries it. + let hi = selected + 1; + let used = rowsAboveSelected + (rowCounts[selected] ?? 0); + while (hi < n && used + (rowCounts[hi] ?? 0) <= budget) { + used += rowCounts[hi] ?? 0; + hi++; + } + + // Step 3: if room remains (selection sat near the bottom), keep + // expanding upward. + while (lo > 0 && used + (rowCounts[lo - 1] ?? 0) <= budget) { + lo--; + used += rowCounts[lo] ?? 0; + } + + let visualOffset = 0; + for (let i = 0; i < lo; i++) visualOffset += rowCounts[i] ?? 0; + return { startIndex: lo, endIndex: hi, visualOffset }; + } + #computeItemLayout( item: SelectItem, isSelected: boolean, diff --git a/packages/tui/test/select-list.test.ts b/packages/tui/test/select-list.test.ts index 684b6ba98..21399880c 100644 --- a/packages/tui/test/select-list.test.ts +++ b/packages/tui/test/select-list.test.ts @@ -306,5 +306,70 @@ describe("SelectList", () => { const rendered = list.render(80).join("\n"); expect(rendered).toContain("█"); }); + + it("caps the popup height at maxVisible rows even when items wrap", () => { + // Three matching items, each wraps to ~5 rows, fits within maxVisible=5 + // budget but the popup must NOT grow to 15 rows. + const items = Array.from({ length: 3 }, (_, i) => ({ + value: `v${i}`, + label: `Item ${i}`, + description: longDescription, + })); + const maxVisible = 5; + const list = new SelectList(items, maxVisible, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 32, + wrapDescription: true, + }); + + const rendered = list.render(80); + // Status status line is gated on overflow (#shouldRenderSearchStatus), + // so the picker proper occupies up to `maxVisible` rows. + expect(rendered.length).toBeLessThanOrEqual(maxVisible); + // Scrollbar must appear since visual rows exceed the budget. + expect(rendered.join("\n")).toContain("█"); + }); + + it("keeps the selected item visible when navigation shifts the window past the budget", () => { + const items = Array.from({ length: 4 }, (_, i) => ({ + value: `v${i}`, + label: `Item ${i}`, + description: longDescription, + })); + const maxVisible = 5; + const list = new SelectList(items, maxVisible, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 32, + wrapDescription: true, + }); + + // Down to the last item. + list.handleInput("\x1b[B"); + list.handleInput("\x1b[B"); + list.handleInput("\x1b[B"); + expect(list.getSelectedItem()?.value).toBe("v3"); + + const rendered = list.render(80); + expect(rendered.length).toBeLessThanOrEqual(maxVisible); + // The selected item's label must appear on screen. + expect(rendered.some(row => row.includes("Item 3"))).toBe(true); + }); + + it("clips a single oversize wrapped item so the popup never exceeds maxVisible rows", () => { + const items = [{ value: "huge", label: "huge", description: longDescription }]; + const maxVisible = 3; + const list = new SelectList(items, maxVisible, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 32, + wrapDescription: true, + }); + + const rendered = list.render(80); + expect(rendered.length).toBeLessThanOrEqual(maxVisible); + // Scrollbar reflects the offscreen tail. + expect(rendered.join("\n")).toContain("█"); + // The first wrapped line (with the primary label) is still visible. + expect(rendered.some(row => row.includes("huge"))).toBe(true); + }); }); }); From 79441bc22c8c7b6b19dbb9274efe1e17fc265436 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 06:48:01 +0000 Subject: [PATCH 05/29] style: bun run fix --- packages/tui/src/components/select-list.ts | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index 30633618d..f4407bb53 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -139,9 +139,7 @@ export class SelectList implements Component { rowCounts[i] = 0; continue; } - rowCounts[i] = wrapEnabled - ? this.#computeItemRowCount(item, conservativeRowWidth, primaryColumnWidth) - : 1; + rowCounts[i] = wrapEnabled ? this.#computeItemRowCount(item, conservativeRowWidth, primaryColumnWidth) : 1; visualTotal += rowCounts[i]; } From b48960e63e24034a287c7655653376b3c26899a3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 08:16:45 +0000 Subject: [PATCH 06/29] fix(mcp): resolved windows stdio shims Resolved Windows stdio MCP commands through PATHEXT before spawning so tools installed as .cmd shims launch from bare command configs. Fixes #2174 --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/mcp/transports/stdio.ts | 148 +++++++++++++++++- .../test/mcp-stdio-transport.test.ts | 42 ++++- 3 files changed, 187 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bef9ae4ac..a03794720 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,7 @@ ### Fixed +- Fixed Windows stdio MCP servers launched through PATH shims such as `codegraph.cmd` so bare commands like `codegraph` resolve via `PATHEXT` before spawn ([#2174](https://github.com/can1357/oh-my-pi/issues/2174)). - Fixed a turn-ending provider error (e.g. a 502 whose body is the proxy's full HTML page) flooding the transcript: `AnthropicApiError` folds the entire response body into `errorMessage`, and the inline transcript render reprinted it verbatim — every embedded blank line included — leaving a tall mostly-empty block ending in ``. The inline error now drops blank lines, clamps to 8 lines, and width-truncates each line via `getPreviewLines`, matching the pinned error banner. ## [15.10.7] - 2026-06-08 diff --git a/packages/coding-agent/src/mcp/transports/stdio.ts b/packages/coding-agent/src/mcp/transports/stdio.ts index 6065d487f..9c4d3a342 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.ts @@ -5,6 +5,9 @@ * Messages are newline-delimited JSON. */ +import * as fs from "node:fs/promises"; +import * as path from "node:path"; + import { getProjectDir, readJsonl, Snowflake } from "@oh-my-pi/pi-utils"; import { type Subprocess, spawn } from "bun"; import type { @@ -19,6 +22,140 @@ import type { import { toJsonRpcError } from "../../mcp/types"; import { isMCPTimeoutEnabled, resolveMCPTimeoutMs } from "../timeout"; +/** Subprocess argv for launching an MCP stdio server. */ +export interface StdioSpawnCommand { + cmd: string[]; +} + +/** Inputs used to resolve platform-specific stdio spawn behavior. */ +export interface ResolveStdioSpawnOptions { + cwd: string; + env: Record; + platform?: NodeJS.Platform; +} + +const DEFAULT_WINDOWS_PATHEXT = [".COM", ".EXE", ".BAT", ".CMD"]; +const WINDOWS_BATCH_EXTENSIONS = new Set([".bat", ".cmd"]); + +function getCaseInsensitiveEnv(env: Record, name: string): string | undefined { + const direct = env[name]; + if (direct !== undefined) return direct; + const normalized = name.toLowerCase(); + for (const [key, value] of Object.entries(env)) { + if (key.toLowerCase() === normalized) return value; + } + return undefined; +} + +function getWindowsPathExt(env: Record): string[] { + const raw = getCaseInsensitiveEnv(env, "PATHEXT"); + if (!raw) return DEFAULT_WINDOWS_PATHEXT; + const extensions: string[] = []; + for (const part of raw.split(";")) { + const trimmed = part.trim(); + if (!trimmed) continue; + extensions.push(trimmed.startsWith(".") ? trimmed : `.${trimmed}`); + } + return extensions.length > 0 ? extensions : DEFAULT_WINDOWS_PATHEXT; +} + +async function fileExists(filePath: string): Promise { + try { + await fs.access(filePath); + return true; + } catch { + return false; + } +} + +function hasPathSegment(command: string): boolean { + return command.includes("/") || command.includes("\\") || path.isAbsolute(command); +} + +function hasExecutableExtension(command: string, extensions: string[]): boolean { + const ext = path.extname(command).toLowerCase(); + if (!ext) return false; + return extensions.some(candidate => candidate.toLowerCase() === ext); +} + +async function resolveWindowsCommandPath( + command: string, + cwd: string, + env: Record, +): Promise { + const extensions = getWindowsPathExt(env); + if (hasExecutableExtension(command, extensions)) return command; + + const candidates = extensions.map(ext => `${command}${ext}`); + if (hasPathSegment(command)) { + for (const candidate of candidates) { + const resolved = path.isAbsolute(candidate) ? candidate : path.resolve(cwd, candidate); + if (await fileExists(resolved)) return resolved; + } + return null; + } + + const pathValue = getCaseInsensitiveEnv(env, "PATH"); + if (!pathValue) return null; + for (const dir of pathValue.split(";")) { + if (!dir) continue; + for (const candidate of candidates) { + const resolved = path.join(dir, candidate); + if (await fileExists(resolved)) return resolved; + } + } + return null; +} + +function quoteCmdArg(value: string): string { + if (value.length === 0) return '""'; + let result = '"'; + let backslashes = 0; + for (const char of value) { + if (char === "\\") { + backslashes++; + continue; + } + if (char === '"') { + result += "\\".repeat(backslashes * 2 + 1); + result += '"'; + backslashes = 0; + continue; + } + result += "\\".repeat(backslashes); + backslashes = 0; + result += char; + } + result += "\\".repeat(backslashes * 2); + return `${result}"`; +} + +function isWindowsBatchCommand(command: string): boolean { + return WINDOWS_BATCH_EXTENSIONS.has(path.extname(command).toLowerCase()); +} + +function resolveComSpec(env: Record): string { + const comspec = getCaseInsensitiveEnv(env, "COMSPEC"); + return comspec && comspec.length > 0 ? comspec : "cmd.exe"; +} + +/** Resolve the subprocess argv used to launch an MCP stdio server. */ +export async function resolveStdioSpawnCommand( + config: MCPStdioServerConfig, + options: ResolveStdioSpawnOptions, +): Promise { + const args = config.args ?? []; + if (options.platform !== "win32") return { cmd: [config.command, ...args] }; + + const resolvedCommand = + (await resolveWindowsCommandPath(config.command, options.cwd, options.env)) ?? config.command; + if (!isWindowsBatchCommand(resolvedCommand)) return { cmd: [resolvedCommand, ...args] }; + + return { + cmd: [resolveComSpec(options.env), "/d", "/s", "/c", [resolvedCommand, ...args].map(quoteCmdArg).join(" ")], + }; +} + /** Minimal write surface of `Subprocess.stdin` we need for framed sends. */ interface FrameSink { write(chunk: string): unknown; @@ -100,15 +237,20 @@ export class StdioTransport implements MCPTransport { async connect(): Promise { if (this.#connected) return; - const args = this.config.args ?? []; const env = { ...Bun.env, ...this.config.env, }; + const cwd = this.config.cwd ?? getProjectDir(); + const spawnCommand = await resolveStdioSpawnCommand(this.config, { + cwd, + env, + platform: process.platform, + }); this.#process = spawn({ - cmd: [this.config.command, ...args], - cwd: this.config.cwd ?? getProjectDir(), + cmd: spawnCommand.cmd, + cwd, env, stdin: "pipe", stdout: "pipe", diff --git a/packages/coding-agent/test/mcp-stdio-transport.test.ts b/packages/coding-agent/test/mcp-stdio-transport.test.ts index 867d9f9cd..2193c6ad7 100644 --- a/packages/coding-agent/test/mcp-stdio-transport.test.ts +++ b/packages/coding-agent/test/mcp-stdio-transport.test.ts @@ -1,5 +1,45 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { StdioTransport, writeFrame } from "@oh-my-pi/pi-coding-agent/mcp/transports/stdio"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; + +import { resolveStdioSpawnCommand, StdioTransport, writeFrame } from "@oh-my-pi/pi-coding-agent/mcp/transports/stdio"; + +describe("resolveStdioSpawnCommand", () => { + it("resolves bare Windows commands through PATHEXT and wraps .cmd shims with cmd.exe", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-mcp-stdio-")); + try { + const shim = path.join(tempDir, "codegraph.cmd"); + await Bun.write(shim, "@echo off\r\n"); + + const result = await resolveStdioSpawnCommand( + { type: "stdio", command: "codegraph", args: ["serve", "--mcp"] }, + { + cwd: tempDir, + env: { + COMSPEC: "C:\\Windows\\System32\\cmd.exe", + PATH: tempDir, + PATHEXT: ".cmd", + }, + platform: "win32", + }, + ); + + expect(result.cmd).toEqual(["C:\\Windows\\System32\\cmd.exe", "/d", "/s", "/c", `"${shim}" "serve" "--mcp"`]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("leaves non-Windows commands untouched", async () => { + const result = await resolveStdioSpawnCommand( + { type: "stdio", command: "codegraph", args: ["serve", "--mcp"] }, + { cwd: "/", env: {}, platform: "linux" }, + ); + + expect(result.cmd).toEqual(["codegraph", "serve", "--mcp"]); + }); +}); // --------------------------------------------------------------------------- // writeFrame — the seam that catches synchronous FileSink throws AND neutralizes From 8201a7200dcd5651347a1460f9ed5652904263c4 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 08:22:11 +0000 Subject: [PATCH 07/29] test(mcp): covered absolute windows shim resolution Added a regression test confirming the stdio resolver promotes extension-less absolute Windows paths (npm-installed shim layout) to their .cmd sibling before launch. Refs #2174 --- .../test/mcp-stdio-transport.test.ts | 33 +++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/packages/coding-agent/test/mcp-stdio-transport.test.ts b/packages/coding-agent/test/mcp-stdio-transport.test.ts index 2193c6ad7..77b9a5c1d 100644 --- a/packages/coding-agent/test/mcp-stdio-transport.test.ts +++ b/packages/coding-agent/test/mcp-stdio-transport.test.ts @@ -31,6 +31,39 @@ describe("resolveStdioSpawnCommand", () => { } }); + it("resolves extension-less absolute Windows paths to the sibling .cmd shim", async () => { + // Mirrors npm's Windows shim layout: bare `codegraph` (shebang script), + // `codegraph.cmd` (cmd.exe wrapper), and `codegraph.ps1` siblings under + // %AppData%\Roaming\npm. uv_spawn rejects the extensionless script; + // the resolver must promote the bare absolute path to its `.cmd` + // sibling so the launch succeeds (see #2174). The test rig pins + // PATHEXT to a single lowercase extension so the candidate filename + // matches the file we create on the case-sensitive test host. + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-mcp-abs-")); + try { + const bare = path.join(tempDir, "codegraph"); + const shim = `${bare}.cmd`; + await Bun.write(bare, "#!/bin/sh\n"); + await Bun.write(shim, "@echo off\r\n"); + + const result = await resolveStdioSpawnCommand( + { type: "stdio", command: bare, args: ["serve", "--mcp"] }, + { + cwd: tempDir, + env: { + COMSPEC: "C:\\Windows\\System32\\cmd.exe", + PATHEXT: ".cmd", + }, + platform: "win32", + }, + ); + + expect(result.cmd).toEqual(["C:\\Windows\\System32\\cmd.exe", "/d", "/s", "/c", `"${shim}" "serve" "--mcp"`]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("leaves non-Windows commands untouched", async () => { const result = await resolveStdioSpawnCommand( { type: "stdio", command: "codegraph", args: ["serve", "--mcp"] }, From d6a04f152f300559ece255a4596fa1b2d61d62c5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 08:23:39 +0000 Subject: [PATCH 08/29] docs(changelog): moved mcp fix to unreleased Moved the Windows stdio MCP changelog note out of the released 15.10.8 section and into Unreleased. Refs #2174 --- packages/coding-agent/CHANGELOG.md | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a03794720..8401e586c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Windows stdio MCP servers launched through PATH shims such as `codegraph.cmd` so bare commands like `codegraph` resolve via `PATHEXT` before spawn ([#2174](https://github.com/can1357/oh-my-pi/issues/2174)). + ## [15.10.8] - 2026-06-09 ### Added @@ -13,7 +17,6 @@ ### Fixed -- Fixed Windows stdio MCP servers launched through PATH shims such as `codegraph.cmd` so bare commands like `codegraph` resolve via `PATHEXT` before spawn ([#2174](https://github.com/can1357/oh-my-pi/issues/2174)). - Fixed a turn-ending provider error (e.g. a 502 whose body is the proxy's full HTML page) flooding the transcript: `AnthropicApiError` folds the entire response body into `errorMessage`, and the inline transcript render reprinted it verbatim — every embedded blank line included — leaving a tall mostly-empty block ending in ``. The inline error now drops blank lines, clamps to 8 lines, and width-truncates each line via `getPreviewLines`, matching the pinned error banner. ## [15.10.7] - 2026-06-08 From 6b4bcb81e4cc6c30a0cd5975731ff245bc718668 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 08:30:16 +0000 Subject: [PATCH 09/29] fix(mcp): preserved percent args in cmd shims Escaped literal percent signs before joining Windows cmd.exe shim command strings so MCP server args are not consumed by cmd environment expansion. Refs #2174 --- .../coding-agent/src/mcp/transports/stdio.ts | 2 +- .../test/mcp-stdio-transport.test.ts | 31 +++++++++++++++++++ 2 files changed, 32 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/mcp/transports/stdio.ts b/packages/coding-agent/src/mcp/transports/stdio.ts index 9c4d3a342..d6e6ebe60 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.ts @@ -124,7 +124,7 @@ function quoteCmdArg(value: string): string { } result += "\\".repeat(backslashes); backslashes = 0; - result += char; + result += char === "%" ? "^%" : char; } result += "\\".repeat(backslashes * 2); return `${result}"`; diff --git a/packages/coding-agent/test/mcp-stdio-transport.test.ts b/packages/coding-agent/test/mcp-stdio-transport.test.ts index 77b9a5c1d..ffb2159c1 100644 --- a/packages/coding-agent/test/mcp-stdio-transport.test.ts +++ b/packages/coding-agent/test/mcp-stdio-transport.test.ts @@ -31,6 +31,37 @@ describe("resolveStdioSpawnCommand", () => { } }); + it("escapes percent-delimited args before routing .cmd shims through cmd.exe", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-mcp-percent-")); + try { + const shim = path.join(tempDir, "codegraph.cmd"); + await Bun.write(shim, "@echo off\r\n"); + + const result = await resolveStdioSpawnCommand( + { type: "stdio", command: "codegraph", args: ["serve", "--header", "Authorization=%TOKEN%"] }, + { + cwd: tempDir, + env: { + COMSPEC: "C:\\Windows\\System32\\cmd.exe", + PATH: tempDir, + PATHEXT: ".cmd", + }, + platform: "win32", + }, + ); + + expect(result.cmd).toEqual([ + "C:\\Windows\\System32\\cmd.exe", + "/d", + "/s", + "/c", + `"${shim}" "serve" "--header" "Authorization=^%TOKEN^%"`, + ]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("resolves extension-less absolute Windows paths to the sibling .cmd shim", async () => { // Mirrors npm's Windows shim layout: bare `codegraph` (shebang script), // `codegraph.cmd` (cmd.exe wrapper), and `codegraph.ps1` siblings under From db351abe492f1a97fbcc0de159bfe68ebe10dc19 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 08:35:44 +0000 Subject: [PATCH 10/29] fix(mcp): escaped cmd shim quotes Escaped literal double quotes with cmd caret syntax before invoking Windows .cmd MCP shims so JSON args cannot break out of the quoted argument. Refs #2174 --- .../coding-agent/src/mcp/transports/stdio.ts | 20 +++++------- .../test/mcp-stdio-transport.test.ts | 31 +++++++++++++++++++ 2 files changed, 38 insertions(+), 13 deletions(-) diff --git a/packages/coding-agent/src/mcp/transports/stdio.ts b/packages/coding-agent/src/mcp/transports/stdio.ts index d6e6ebe60..d3d0e2212 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.ts @@ -110,23 +110,17 @@ async function resolveWindowsCommandPath( function quoteCmdArg(value: string): string { if (value.length === 0) return '""'; let result = '"'; - let backslashes = 0; for (const char of value) { - if (char === "\\") { - backslashes++; - continue; - } if (char === '"') { - result += "\\".repeat(backslashes * 2 + 1); - result += '"'; - backslashes = 0; - continue; + result += '^"'; + } else if (char === "^") { + result += "^^"; + } else if (char === "%") { + result += "^%"; + } else { + result += char; } - result += "\\".repeat(backslashes); - backslashes = 0; - result += char === "%" ? "^%" : char; } - result += "\\".repeat(backslashes * 2); return `${result}"`; } diff --git a/packages/coding-agent/test/mcp-stdio-transport.test.ts b/packages/coding-agent/test/mcp-stdio-transport.test.ts index ffb2159c1..5f3396c74 100644 --- a/packages/coding-agent/test/mcp-stdio-transport.test.ts +++ b/packages/coding-agent/test/mcp-stdio-transport.test.ts @@ -62,6 +62,37 @@ describe("resolveStdioSpawnCommand", () => { } }); + it("escapes quoted JSON args before routing .cmd shims through cmd.exe", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-mcp-quotes-")); + try { + const shim = path.join(tempDir, "codegraph.cmd"); + await Bun.write(shim, "@echo off\r\n"); + + const result = await resolveStdioSpawnCommand( + { type: "stdio", command: "codegraph", args: ["--config", '{"a":"b&c|d"}'] }, + { + cwd: tempDir, + env: { + COMSPEC: "C:\\Windows\\System32\\cmd.exe", + PATH: tempDir, + PATHEXT: ".cmd", + }, + platform: "win32", + }, + ); + + expect(result.cmd).toEqual([ + "C:\\Windows\\System32\\cmd.exe", + "/d", + "/s", + "/c", + `"${shim}" "--config" "{^"a^":^"b&c|d^"}"`, + ]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("resolves extension-less absolute Windows paths to the sibling .cmd shim", async () => { // Mirrors npm's Windows shim layout: bare `codegraph` (shebang script), // `codegraph.cmd` (cmd.exe wrapper), and `codegraph.ps1` siblings under From f5f4f2265b632f7bcccf4987d0b504c018d0d6fc Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 09:16:41 +0000 Subject: [PATCH 11/29] fix(ai): widen first-event watchdog for DeepSeek V4 reasoning MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DeepSeek V4 reasoning models on api.deepseek.com emit no SSE bytes while the model finishes its private chain-of-thought, which routinely takes longer than the generic 100s first-event budget under load. The OpenAI completions stream then aborts with "OpenAI completions stream timed out while waiting for the first event" and silently retries, doubling user-visible latency on almost every chat. getOpenAICompletionsStreamIdleTimeoutFallbackMs now returns a 300s floor for reasoning models when (provider === "deepseek") or (baseUrl includes api.deepseek.com). first-event floors at idle, so the watchdog gains 5 minutes — enough for reasoning warm-ups — without changing steady-state streaming behavior. Mirrors the existing GLM coding-plan widening. Fixes #2177 --- packages/ai/CHANGELOG.md | 4 ++ .../ai/src/providers/openai-completions.ts | 33 +++++++++--- .../openai-completions-progress-chunk.test.ts | 52 +++++++++++++++++++ 3 files changed, 82 insertions(+), 7 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 16e224a18..88f20073a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Widened the OpenAI-completions first-event watchdog floor from 120s to 300s for DeepSeek V4 reasoning models hosted on the official DeepSeek API. The reasoner emits no SSE bytes until its private chain-of-thought finishes, which routinely takes longer than the generic 100s first-event budget under load — every chat then aborted with `OpenAI completions stream timed out while waiting for the first event` and silently retried. Mirrors the existing GLM coding-plan widening ([#2177](https://github.com/can1357/oh-my-pi/issues/2177)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 808077012..4072174c0 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -392,17 +392,36 @@ const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE = const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i; -/** Returns the widened OpenAI stream watchdog floor for slow GLM coding-plan reasoning models. */ +// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE +// bytes while the model finishes its private chain-of-thought, which routinely +// takes longer than the generic 100s first-event floor under load (issue +// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the +// first-event watchdog (it floors at idle) without changing the runtime +// streaming behavior, so reasoning warm-ups stop aborting and retrying. +const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; + +function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean { + if (!model.reasoning) return false; + if (model.provider === "deepseek") return true; + return model.baseUrl.toLowerCase().includes("api.deepseek.com"); +} + +/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */ export function getOpenAICompletionsStreamIdleTimeoutFallbackMs( model: Model<"openai-completions">, ): number | undefined { - if (!GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) return undefined; - if (model.provider === "zhipu-coding-plan" || model.provider === "zai") - return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) { + if (model.provider === "zhipu-coding-plan" || model.provider === "zai") + return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; - const baseUrl = model.baseUrl.toLowerCase(); - if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) { - return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + const baseUrl = model.baseUrl.toLowerCase(); + if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) { + return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + } + } + + if (isDirectDeepseekReasoningModel(model)) { + return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS; } return undefined; diff --git a/packages/ai/test/openai-completions-progress-chunk.test.ts b/packages/ai/test/openai-completions-progress-chunk.test.ts index 7e92a8990..933f80717 100644 --- a/packages/ai/test/openai-completions-progress-chunk.test.ts +++ b/packages/ai/test/openai-completions-progress-chunk.test.ts @@ -103,6 +103,58 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000); }); + it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-v4-pro", + name: "DeepSeek V4 Pro", + provider: "deepseek", + baseUrl: "https://api.deepseek.com", + reasoning: true, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000); + }); + + it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-v4-pro", + name: "DeepSeek V4 Pro", + provider: "openai", + baseUrl: "https://api.deepseek.com/v1", + reasoning: true, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000); + }); + + it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-chat", + name: "DeepSeek Chat", + provider: "deepseek", + baseUrl: "https://api.deepseek.com", + reasoning: false, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined(); + }); + + it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-v4-pro", + name: "DeepSeek V4 Pro", + provider: "aimlapi", + baseUrl: "https://api.aimlapi.com/v1", + reasoning: true, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined(); + }); + it("keeps ordinary OpenAI-compatible models on the global timeout", () => { expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined(); }); From daee87f00e18386f5de822b0e2b486a2cf484385 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 09:46:52 +0000 Subject: [PATCH 12/29] fix(ssh): cancelled controlmaster stream waits Return promptly when the ssh executor receives a user interrupt while ControlMaster-owned streams remain open. Fixes #2180 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/ssh/ssh-executor.ts | 61 +++++++++++++++-- .../test/ssh/ssh-executor.test.ts | 65 +++++++++++++++++++ 3 files changed, 126 insertions(+), 4 deletions(-) create mode 100644 packages/coding-agent/test/ssh/ssh-executor.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bef9ae4ac..c17372cb0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed SSH tool cancellation hanging behind OpenSSH ControlMaster streams that stayed open after an Esc/user interrupt ([#2180](https://github.com/can1357/oh-my-pi/issues/2180)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/coding-agent/src/ssh/ssh-executor.ts b/packages/coding-agent/src/ssh/ssh-executor.ts index 3dd6f5b6d..6e3a7be28 100644 --- a/packages/coding-agent/src/ssh/ssh-executor.ts +++ b/packages/coding-agent/src/ssh/ssh-executor.ts @@ -42,6 +42,42 @@ export interface SSHResult { artifactId?: string; } +type SSHExitEvent = { kind: "exit"; exitCode: number } | { kind: "error"; error: unknown }; + +function sshExitEvent(exitCode: number): SSHExitEvent { + return { kind: "exit", exitCode }; +} + +function sshErrorEvent(error: unknown): SSHExitEvent { + return { kind: "error", error }; +} + +function createAbortWaiter( + signal: AbortSignal | undefined, + streamAbort: AbortController, +): { promise: Promise | undefined; cleanup: () => void } { + if (!signal) { + return { promise: undefined, cleanup: () => {} }; + } + + const { promise, resolve } = Promise.withResolvers(); + const onAbort = () => { + const error = new ptree.AbortError(signal.reason, ""); + if (!streamAbort.signal.aborted) { + streamAbort.abort(error); + } + resolve(error); + }; + + if (signal.aborted) { + onAbort(); + return { promise, cleanup: () => {} }; + } + + signal.addEventListener("abort", onAbort, { once: true }); + return { promise, cleanup: () => signal.removeEventListener("abort", onAbort) }; +} + function quoteForCompatShell(command: string): string { if (command.length === 0) { return "''"; @@ -94,19 +130,34 @@ export async function executeSSH( maxColumns: resolveOutputMaxColumns(settings), }); - const streams = [child.stdout.pipeTo(sink.createInput())]; + const streamAbort = new AbortController(); + const abortWaiter = createAbortWaiter(options?.signal, streamAbort); + const streamOptions = { signal: streamAbort.signal }; + const streams = [child.stdout.pipeTo(sink.createInput(), streamOptions)]; if (child.stderr) { - streams.push(child.stderr.pipeTo(sink.createInput())); + streams.push(child.stderr.pipeTo(sink.createInput(), streamOptions)); } - await Promise.allSettled(streams).catch(() => {}); + const streamsSettled = Promise.allSettled(streams).then(() => {}); try { + const exitEvent = child.exited.then(sshExitEvent, sshErrorEvent); + const abortEvent = abortWaiter.promise?.then(sshErrorEvent); + const event = await (abortEvent ? Promise.race([exitEvent, abortEvent]) : exitEvent); + if (event.kind === "error") { + throw event.error; + } + + await streamsSettled; return { - exitCode: await child.exited, + exitCode: event.exitCode, cancelled: false, ...(await sink.dump()), }; } catch (err) { + if (!streamAbort.signal.aborted) { + streamAbort.abort(err); + } + void streamsSettled; if (err instanceof ptree.Exception) { if (err instanceof ptree.TimeoutError) { return { @@ -129,5 +180,7 @@ export async function executeSSH( }; } throw err; + } finally { + abortWaiter.cleanup(); } } diff --git a/packages/coding-agent/test/ssh/ssh-executor.test.ts b/packages/coding-agent/test/ssh/ssh-executor.test.ts new file mode 100644 index 000000000..085c32531 --- /dev/null +++ b/packages/coding-agent/test/ssh/ssh-executor.test.ts @@ -0,0 +1,65 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import * as connectionManager from "@oh-my-pi/pi-coding-agent/ssh/connection-manager"; +import { executeSSH } from "@oh-my-pi/pi-coding-agent/ssh/ssh-executor"; +import * as sshfsMount from "@oh-my-pi/pi-coding-agent/ssh/sshfs-mount"; +import { type ChildProcess, ptree } from "@oh-my-pi/pi-utils"; + +type TestStdin = "pipe" | "ignore" | Buffer | Uint8Array | null; + +function createNeverClosingStream(): ReadableStream { + return new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode("started\n")); + }, + }); +} + +function createBlockedChild(): ChildProcess { + const { promise } = Promise.withResolvers(); + + return { + stdout: createNeverClosingStream(), + stderr: undefined, + exited: promise, + [Symbol.dispose]() {}, + } as unknown as ChildProcess; +} + +async function flushMicrotasks(count: number): Promise { + for (let i = 0; i < count; i++) { + await Promise.resolve(); + } +} + +describe("executeSSH", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("returns promptly when an abort races a ControlMaster stream that stays open", async () => { + vi.spyOn(connectionManager, "ensureConnection").mockResolvedValue(); + vi.spyOn(connectionManager, "buildRemoteCommand").mockResolvedValue(["remote", "sleep 60"]); + vi.spyOn(sshfsMount, "hasSshfs").mockReturnValue(false); + vi.spyOn(ptree, "spawn").mockImplementation(() => createBlockedChild()); + + const chunked = Promise.withResolvers(); + const controller = new AbortController(); + const resultPromise = executeSSH({ name: "remote", host: "remote" }, "sleep 60", { + signal: controller.signal, + onChunk: () => chunked.resolve(), + }); + await chunked.promise; + + let result: Awaited | undefined; + resultPromise.then(value => { + result = value; + }); + controller.abort("user interrupt"); + await flushMicrotasks(20); + expect(result).toBeDefined(); + if (!result) return; + expect(result.cancelled).toBe(true); + expect(result.exitCode).toBeUndefined(); + expect(result.output).toContain("Command aborted"); + }); +}); From 80a47a83291e383d4bf01aee687c7e80dcffb8fe Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 09:54:15 +0000 Subject: [PATCH 13/29] fix(ssh): report abort after process exit Preserve cancellation semantics when a user interrupt unblocks SSH stream drains after the mux client already exited. Fixes #2180 --- packages/coding-agent/src/ssh/ssh-executor.ts | 5 ++- .../test/ssh/ssh-executor.test.ts | 42 ++++++++++++++++--- 2 files changed, 40 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/ssh/ssh-executor.ts b/packages/coding-agent/src/ssh/ssh-executor.ts index 6e3a7be28..dbe553cd6 100644 --- a/packages/coding-agent/src/ssh/ssh-executor.ts +++ b/packages/coding-agent/src/ssh/ssh-executor.ts @@ -147,7 +147,10 @@ export async function executeSSH( throw event.error; } - await streamsSettled; + const streamEvent = await (abortEvent ? Promise.race([streamsSettled, abortEvent]) : streamsSettled); + if (streamEvent?.kind === "error") { + throw streamEvent.error; + } return { exitCode: event.exitCode, cancelled: false, diff --git a/packages/coding-agent/test/ssh/ssh-executor.test.ts b/packages/coding-agent/test/ssh/ssh-executor.test.ts index 085c32531..d261ff157 100644 --- a/packages/coding-agent/test/ssh/ssh-executor.test.ts +++ b/packages/coding-agent/test/ssh/ssh-executor.test.ts @@ -14,13 +14,13 @@ function createNeverClosingStream(): ReadableStream { }); } -function createBlockedChild(): ChildProcess { +function createBlockedChild(exited?: Promise): ChildProcess { const { promise } = Promise.withResolvers(); return { stdout: createNeverClosingStream(), stderr: undefined, - exited: promise, + exited: exited ?? promise, [Symbol.dispose]() {}, } as unknown as ChildProcess; } @@ -36,19 +36,49 @@ describe("executeSSH", () => { vi.restoreAllMocks(); }); - it("returns promptly when an abort races a ControlMaster stream that stays open", async () => { + function mockOpenStreamChild(exited?: Promise) { vi.spyOn(connectionManager, "ensureConnection").mockResolvedValue(); vi.spyOn(connectionManager, "buildRemoteCommand").mockResolvedValue(["remote", "sleep 60"]); vi.spyOn(sshfsMount, "hasSshfs").mockReturnValue(false); - vi.spyOn(ptree, "spawn").mockImplementation(() => createBlockedChild()); + vi.spyOn(ptree, "spawn").mockImplementation(() => createBlockedChild(exited)); + } + function startOpenStreamCommand(controller: AbortController) { const chunked = Promise.withResolvers(); - const controller = new AbortController(); const resultPromise = executeSSH({ name: "remote", host: "remote" }, "sleep 60", { signal: controller.signal, onChunk: () => chunked.resolve(), }); - await chunked.promise; + return { resultPromise, chunked: chunked.promise }; + } + + it("returns promptly when an abort races a ControlMaster stream that stays open", async () => { + mockOpenStreamChild(); + + const controller = new AbortController(); + const { resultPromise, chunked } = startOpenStreamCommand(controller); + await chunked; + + let result: Awaited | undefined; + resultPromise.then(value => { + result = value; + }); + controller.abort("user interrupt"); + await flushMicrotasks(20); + expect(result).toBeDefined(); + if (!result) return; + expect(result.cancelled).toBe(true); + expect(result.exitCode).toBeUndefined(); + expect(result.output).toContain("Command aborted"); + }); + + it("reports cancellation when abort unblocks streams after the ssh process exits", async () => { + mockOpenStreamChild(Promise.resolve(0)); + + const controller = new AbortController(); + const { resultPromise, chunked } = startOpenStreamCommand(controller); + await chunked; + await flushMicrotasks(20); let result: Awaited | undefined; resultPromise.then(value => { From e64aa626825cfaee6438d01d98a223973b3a96b7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 13:00:42 +0000 Subject: [PATCH 14/29] fix(ai): rotated google-antigravity accounts when capacity exhausted MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cloud Code Assist's quota-exhaustion 429 carries 'You have exhausted your capacity on this model. Your quota will reset after …'. The literal 'capacity' hit the MODEL_CAPACITY_EXHAUSTED branch in parseRateLimitReason before the 'quota will reset' suffix had a say, downgrading the failure to a 45-75s transient backoff; isUsageLimitError missed the same shape, so agent-session.ts (line 8314) and auth-storage.ts (line 3457) never invoked markUsageLimitReached. The session kept hammering the exhausted credential while the retry layer bailed on the multi-hour retry-after, even when sibling Antigravity OAuth accounts had headroom. - Short-circuit parseRateLimitReason for 'quota will reset' / 'exhausted your capacity' to QUOTA_EXHAUSTED before the MODEL_CAPACITY fallthrough. - Extend USAGE_LIMIT_PATTERN so isUsageLimitError matches the Antigravity phrasing too, enabling credential-rotation paths in agent-session, auth-retry, stream, and the auth-gateway. - Add antigravityRankingStrategy and register it under DEFAULT_RANKING_STRATEGIES so multi-account selection consumes the per-counter Antigravity usage report (already sorted ascending by remainingFraction in fetchAntigravityUsage) instead of falling back to round-robin. - Cover the regression with parseRateLimitReason / isUsageLimitError tests on the literal Antigravity message, plus a ranking-strategy contract test asserting primary/secondary mapping and the 24h fallback window. Fixes #2187 --- packages/ai/CHANGELOG.md | 8 +++ packages/ai/src/auth-storage.ts | 3 +- packages/ai/src/rate-limit-utils.ts | 16 ++++- packages/ai/src/usage/google-antigravity.ts | 32 ++++++++++ .../ai/test/google-antigravity-usage.test.ts | 59 ++++++++++++++++++- packages/ai/test/rate-limit-utils.test.ts | 24 ++++++++ 6 files changed, 136 insertions(+), 6 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 16e224a18..0b86b6cd8 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Fixed + +- Fixed `google-antigravity` not rotating to another stored OAuth account when Cloud Code Assist returns `429 You have exhausted your capacity on this model. Your quota will reset after …`. `parseRateLimitReason` matched the literal `capacity` before the `quota will reset` suffix and downgraded the failure to `MODEL_CAPACITY_EXHAUSTED` (45–75 s backoff), and `isUsageLimitError` returned false for the same message — so `markUsageLimitReached` was never invoked and the agent kept hammering the exhausted credential while the retry layer bailed on the multi-hour `retry-after`. Both paths now treat the Antigravity phrasing as `QUOTA_EXHAUSTED` / usage-limit, blocking the current credential until reset and letting the session pick an unblocked sibling ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). + +### Added + +- Added `antigravityRankingStrategy` and registered it as the default `CredentialRankingStrategy` for `google-antigravity`, so multi-account selection consumes the per-counter Antigravity usage reports (sorted ascending by `remainingFraction` in `fetchAntigravityUsage`) before falling back to round-robin — preventing the exhausted-counter credential from being chosen first when an unblocked sibling has headroom ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 2c1a96958..482d0d65c 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -31,7 +31,7 @@ import type { import { claudeRankingStrategy, claudeUsageProvider } from "./usage/claude"; import { googleGeminiCliUsageProvider } from "./usage/gemini"; import { githubCopilotUsageProvider } from "./usage/github-copilot"; -import { antigravityUsageProvider } from "./usage/google-antigravity"; +import { antigravityRankingStrategy, antigravityUsageProvider } from "./usage/google-antigravity"; import { kimiUsageProvider } from "./usage/kimi"; import { codexRankingStrategy, openaiCodexUsageProvider } from "./usage/openai-codex"; import { zaiUsageProvider } from "./usage/zai"; @@ -650,6 +650,7 @@ function resolveDefaultUsageProvider(provider: Provider): UsageProvider | undefi const DEFAULT_RANKING_STRATEGIES = new Map([ ["openai-codex", codexRankingStrategy], ["anthropic", claudeRankingStrategy], + ["google-antigravity", antigravityRankingStrategy], ]); function resolveDefaultRankingStrategy(provider: Provider): CredentialRankingStrategy | undefined { diff --git a/packages/ai/src/rate-limit-utils.ts b/packages/ai/src/rate-limit-utils.ts index bea194914..84db0cb1b 100644 --- a/packages/ai/src/rate-limit-utils.ts +++ b/packages/ai/src/rate-limit-utils.ts @@ -21,14 +21,24 @@ const ACCOUNT_RATE_LIMIT_PATTERN = /** * Classify a rate-limit error message into a reason category. - * Priority order: MODEL_CAPACITY > RATE_LIMIT > QUOTA > SERVER_ERROR > UNKNOWN. + * Priority order: QUOTA (Antigravity "quota will reset") > MODEL_CAPACITY > QUOTA (account) > + * RATE_LIMIT > QUOTA (generic) > SERVER_ERROR > UNKNOWN. * * "resource exhausted" maps to MODEL_CAPACITY (transient, short wait) - * "quota exceeded" maps to QUOTA_EXHAUSTED (long wait, switch account) + * "quota exceeded" / "quota will reset" maps to QUOTA_EXHAUSTED (long wait, switch account) */ export function parseRateLimitReason(errorMessage: string): RateLimitReason { const lower = errorMessage.toLowerCase(); + // Antigravity / Cloud Code Assist surface multi-hour daily-quota exhaustion as + // "You have exhausted your capacity on this model. Your quota will reset after …". + // The literal "capacity" used to pre-empt the QUOTA branch even though "quota + // will reset" is the long-wait signal — short-circuit here before the + // MODEL_CAPACITY fallthrough so credential rotation (not 60s backoff) kicks in. + if (lower.includes("quota will reset") || lower.includes("exhausted your capacity")) { + return "QUOTA_EXHAUSTED"; + } + if ( lower.includes("capacity") || lower.includes("overloaded") || @@ -84,7 +94,7 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number { /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */ const USAGE_LIMIT_PATTERN = - /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?exceeded|resource.?exhausted/i; + /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?exceeded|resource.?exhausted|exhausted your capacity|quota will reset/i; export function isUsageLimitError(errorMessage: string): boolean { return USAGE_LIMIT_PATTERN.test(errorMessage) || ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage); diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 215b8b17b..9e8887c5e 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -1,5 +1,6 @@ import { getAntigravityUserAgent } from "../providers/google-gemini-headers"; import type { + CredentialRankingStrategy, UsageAmount, UsageFetchContext, UsageFetchParams, @@ -299,3 +300,34 @@ export const antigravityUsageProvider: UsageProvider = { fetchUsage: fetchAntigravityUsage, supports: params => params.provider === "google-antigravity", }; + +const ANTIGRAVITY_DAILY_WINDOW_MS = 24 * 60 * 60 * 1000; + +/** + * Credential ranking strategy for `google-antigravity`. Drives proactive + * multi-account selection in {@link AuthStorage} by reading the per-counter + * Antigravity usage reports. + * + * Antigravity reports one {@link UsageLimit} per backend counter (Google / + * Anthropic / OpenAI) per tier per window, and {@link fetchAntigravityUsage} + * sorts them ascending by `remainingFraction` — so `limits[0]` is always the + * most-pressured counter for the credential, and `limits[1]` (when present) + * is the next-most-pressured. Mapping those to {primary, secondary} lets the + * ranker compare two credentials on both their bottleneck counter and their + * runner-up before falling back to round-robin. + * + * The Antigravity API exposes `resetTime` but not window duration, so the + * drain-rate calculation depends on `windowDefaults`. Antigravity quotas are + * effectively daily; 24h is the right fallback for both axes — any 5h tier + * still ranks correctly because both credentials are normalised against the + * same fallback. + */ +export const antigravityRankingStrategy: CredentialRankingStrategy = { + findWindowLimits(report) { + return { primary: report.limits[0], secondary: report.limits[1] }; + }, + windowDefaults: { + primaryMs: ANTIGRAVITY_DAILY_WINDOW_MS, + secondaryMs: ANTIGRAVITY_DAILY_WINDOW_MS, + }, +}; diff --git a/packages/ai/test/google-antigravity-usage.test.ts b/packages/ai/test/google-antigravity-usage.test.ts index 2beb9d54a..4a734cb9c 100644 --- a/packages/ai/test/google-antigravity-usage.test.ts +++ b/packages/ai/test/google-antigravity-usage.test.ts @@ -6,8 +6,8 @@ */ import { describe, expect, it } from "bun:test"; import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; -import type { UsageFetchContext, UsageFetchParams } from "@oh-my-pi/pi-ai/usage"; -import { antigravityUsageProvider } from "@oh-my-pi/pi-ai/usage/google-antigravity"; +import type { UsageFetchContext, UsageFetchParams, UsageLimit } from "@oh-my-pi/pi-ai/usage"; +import { antigravityRankingStrategy, antigravityUsageProvider } from "@oh-my-pi/pi-ai/usage/google-antigravity"; const accessTokenFixture = (() => { const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url"); @@ -237,3 +237,58 @@ describe("antigravity usage provider", () => { expect(report).toBeNull(); }); }); + +describe("antigravity ranking strategy", () => { + function makeLimit(remainingFraction: number, label = "Usage"): UsageLimit { + const usedFraction = 1 - remainingFraction; + return { + id: `google-antigravity:${label.toLowerCase()}`, + label, + scope: { provider: "google-antigravity" }, + amount: { + unit: "percent", + remainingFraction, + usedFraction, + remaining: remainingFraction * 100, + used: usedFraction * 100, + limit: 100, + }, + status: remainingFraction <= 0 ? "exhausted" : remainingFraction <= 0.1 ? "warning" : "ok", + }; + } + + it("maps the most-pressured counter to primary and the runner-up to secondary", () => { + // fetchAntigravityUsage sorts ascending by remainingFraction, so a real + // report's limits[0] is always the bottleneck — the ranking strategy + // MUST surface that as the primary signal, not the first model alphabetically. + const report = { + provider: "google-antigravity" as const, + fetchedAt: Date.now(), + limits: [makeLimit(0.05, "Anthropic"), makeLimit(0.4, "Google"), makeLimit(0.9, "OpenAI")], + }; + const { primary, secondary } = antigravityRankingStrategy.findWindowLimits(report); + expect(primary?.label).toBe("Anthropic"); + expect(secondary?.label).toBe("Google"); + }); + + it("returns undefined windows when the credential has no usage limits", () => { + const report = { + provider: "google-antigravity" as const, + fetchedAt: Date.now(), + limits: [], + }; + const { primary, secondary } = antigravityRankingStrategy.findWindowLimits(report); + expect(primary).toBeUndefined(); + expect(secondary).toBeUndefined(); + }); + + it("uses a 24h window default for drain-rate normalisation", () => { + // Antigravity's API exposes resetTime but not durationMs, so AuthStorage's + // drain-rate calculator falls back to windowDefaults. The constant has to + // match the daily quota Antigravity actually applies; if it drifts, two + // credentials with identical headroom but different windowIds will be + // ranked unfairly. + expect(antigravityRankingStrategy.windowDefaults.primaryMs).toBe(24 * 60 * 60 * 1000); + expect(antigravityRankingStrategy.windowDefaults.secondaryMs).toBe(24 * 60 * 60 * 1000); + }); +}); diff --git a/packages/ai/test/rate-limit-utils.test.ts b/packages/ai/test/rate-limit-utils.test.ts index 69406e3f5..44ea85770 100644 --- a/packages/ai/test/rate-limit-utils.test.ts +++ b/packages/ai/test/rate-limit-utils.test.ts @@ -53,6 +53,19 @@ describe("parseRateLimitReason", () => { ), ).toBe("QUOTA_EXHAUSTED"); }); + + it("classifies Antigravity capacity-exhausted as QUOTA_EXHAUSTED, not transient MODEL_CAPACITY", () => { + // Antigravity returns "You have exhausted your capacity on this model. Your + // quota will reset after 3h6m38s." The literal "capacity" used to win the + // classifier race and land in MODEL_CAPACITY_EXHAUSTED (45-75s backoff), + // blocking the agent from rotating to another OAuth account even though the + // "quota will reset" suffix is the long-wait, switch-account signal. + expect( + parseRateLimitReason( + "Cloud Code Assist API error (429): You have exhausted your capacity on this model. Your quota will reset after 3h6m38s.", + ), + ).toBe("QUOTA_EXHAUSTED"); + }); }); describe("isUsageLimitError", () => { @@ -63,6 +76,17 @@ describe("isUsageLimitError", () => { ), ).toBe(true); }); + + it("detects Antigravity capacity-exhausted message as a usage-limit error", () => { + // Without this branch `markUsageLimitReached` is never invoked, so the + // session sticks to the exhausted OAuth account instead of rotating — + // see `agent-session.ts` line 8314 and `auth-storage.ts` line 3457. + expect( + isUsageLimitError( + "Cloud Code Assist API error (429): You have exhausted your capacity on this model. Your quota will reset after 3h6m38s.", + ), + ).toBe(true); + }); }); describe("calculateRateLimitBackoffMs", () => { From c9c2da9c8ab2ebae45bd199680ddb8e6017d2889 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 13:06:00 +0000 Subject: [PATCH 15/29] fix(ai): ranked antigravity bottleneck counters first The google-antigravity ranking strategy originally returned the most-pressured counter as primary and the runner-up as secondary. AuthStorage compares the secondary ranking metrics before primary, so credentials with a dangerous bottleneck but a healthy runner-up counter could outrank credentials with balanced headroom. Return the bottleneck as secondary and the runner-up as primary, and update the strategy contract test to lock the ordering invariant. Fixes #2187 --- packages/ai/src/usage/google-antigravity.ts | 13 +++++++++---- packages/ai/test/google-antigravity-usage.test.ts | 12 +++++++----- 2 files changed, 16 insertions(+), 9 deletions(-) diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 9e8887c5e..435e039a4 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -312,9 +312,14 @@ const ANTIGRAVITY_DAILY_WINDOW_MS = 24 * 60 * 60 * 1000; * Anthropic / OpenAI) per tier per window, and {@link fetchAntigravityUsage} * sorts them ascending by `remainingFraction` — so `limits[0]` is always the * most-pressured counter for the credential, and `limits[1]` (when present) - * is the next-most-pressured. Mapping those to {primary, secondary} lets the - * ranker compare two credentials on both their bottleneck counter and their - * runner-up before falling back to round-robin. + * is the next-most-pressured counter. + * + * `AuthStorage` compares the `secondary*` ranking metrics before `primary*` + * because other providers model a long-window budget as secondary. Antigravity + * does not expose a short/long split; every counter is a sibling bottleneck. + * Therefore the most-pressured counter goes in `secondary`, with the runner-up + * in `primary`, so proactive account selection always ranks the bottleneck + * before any healthier sibling counter. * * The Antigravity API exposes `resetTime` but not window duration, so the * drain-rate calculation depends on `windowDefaults`. Antigravity quotas are @@ -324,7 +329,7 @@ const ANTIGRAVITY_DAILY_WINDOW_MS = 24 * 60 * 60 * 1000; */ export const antigravityRankingStrategy: CredentialRankingStrategy = { findWindowLimits(report) { - return { primary: report.limits[0], secondary: report.limits[1] }; + return { primary: report.limits[1], secondary: report.limits[0] }; }, windowDefaults: { primaryMs: ANTIGRAVITY_DAILY_WINDOW_MS, diff --git a/packages/ai/test/google-antigravity-usage.test.ts b/packages/ai/test/google-antigravity-usage.test.ts index 4a734cb9c..83bbe8b8c 100644 --- a/packages/ai/test/google-antigravity-usage.test.ts +++ b/packages/ai/test/google-antigravity-usage.test.ts @@ -257,18 +257,20 @@ describe("antigravity ranking strategy", () => { }; } - it("maps the most-pressured counter to primary and the runner-up to secondary", () => { + it("maps the most-pressured counter to secondary because AuthStorage compares secondary first", () => { // fetchAntigravityUsage sorts ascending by remainingFraction, so a real - // report's limits[0] is always the bottleneck — the ranking strategy - // MUST surface that as the primary signal, not the first model alphabetically. + // report's limits[0] is always the bottleneck. AuthStorage compares the + // secondary ranking metrics before primary, so Antigravity must put the + // bottleneck there; otherwise [5%, 90%] remaining can beat [40%, 40%] + // because the runner-up counter looks healthier. const report = { provider: "google-antigravity" as const, fetchedAt: Date.now(), limits: [makeLimit(0.05, "Anthropic"), makeLimit(0.4, "Google"), makeLimit(0.9, "OpenAI")], }; const { primary, secondary } = antigravityRankingStrategy.findWindowLimits(report); - expect(primary?.label).toBe("Anthropic"); - expect(secondary?.label).toBe("Google"); + expect(secondary?.label).toBe("Anthropic"); + expect(primary?.label).toBe("Google"); }); it("returns undefined windows when the credential has no usage limits", () => { From e7645cec4902acad952f18a32494a0dd2c199db3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 13:58:21 +0000 Subject: [PATCH 16/29] fix(task): forward parent-discovered rules, extensions, and custom tools to subagents Each `runSubprocess` call re-ran `loadCapability()`, `loadSessionExtensions()`, and `discoverAndLoadCustomTools()` because `ExecutorOptions` and the `createAgentSession()` call inside the executor omitted three pass-through fields the parent had already paid for. The already-correct paths (skills, context files, workspace tree, MCP manager) showed the intended pattern. - Cache `rules`, `extensionsResult`, and `loadedCustomTools` on the parent's `ToolSession`. - Add `rules` / `preloadedExtensions` / `preloadedCustomTools` to `ExecutorOptions`; forward them from both `runSubprocess` call sites in `task/index.ts` and into the executor's `createAgentSession()`. - Add `preloadedCustomTools` to `CreateAgentSessionOptions` and skip `discoverAndLoadCustomTools()` when it is supplied. - Shallow-clone `extensionsResult.extensions` when reusing `preloadedExtensions`, so the per-session autoresearch + custom-tools inline wrappers never leak back into the caller's array. Fixes #2190 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/sdk.ts | 107 +++++++++----- packages/coding-agent/src/task/executor.ts | 13 +- packages/coding-agent/src/task/index.ts | 6 + packages/coding-agent/src/tools/index.ts | 9 ++ ...sdk-preloaded-extensions-isolation.test.ts | 75 ++++++++++ .../test/task/executor-pass-through.test.ts | 139 ++++++++++++++++++ 7 files changed, 315 insertions(+), 38 deletions(-) create mode 100644 packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts create mode 100644 packages/coding-agent/test/task/executor-pass-through.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bef9ae4ac..c9159d46e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `task`-spawned subagents repeating filesystem scans the parent had already completed. `ExecutorOptions` and the `createAgentSession()` call inside `runSubprocess()` did not forward `rules`, `preloadedExtensions`, or the discovered `.omp/tools/` (a.k.a. `LoadedCustomTool[]`) results, so each subagent re-ran `loadCapability()`, `loadSessionExtensions()`, and `discoverAndLoadCustomTools()`. The toolsession now caches each (`session.rules`, `session.extensionsResult`, `session.loadedCustomTools`), `runSubprocess()` threads them through, and `createAgentSession()` accepts a new `preloadedCustomTools` option. `preloadedExtensions` is now shallow-cloned inside the SDK so inline-extension augmentation (autoresearch + custom-tools wrapper) can never bleed back into the caller's array ([#2190](https://github.com/can1357/oh-my-pi/issues/2190)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 75fdb3966..c5db5814a 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -63,7 +63,12 @@ import { loadCustomCommands as loadCustomCommandsInternal, } from "./extensibility/custom-commands"; import { discoverAndLoadCustomTools } from "./extensibility/custom-tools"; -import type { CustomTool, CustomToolContext, CustomToolSessionEvent } from "./extensibility/custom-tools/types"; +import type { + CustomTool, + CustomToolContext, + CustomToolSessionEvent, + LoadedCustomTool, +} from "./extensibility/custom-tools/types"; import { discoverAndLoadExtensions, type ExtensionContext, @@ -341,6 +346,14 @@ export interface CreateAgentSessionOptions { * @internal Used by CLI when extensions are loaded early to parse custom flags. */ preloadedExtensions?: LoadExtensionsResult; + /** + * Pre-loaded custom tools from `.omp/tools/`, `.claude/tools/`, plugins, etc. + * When provided, the filesystem-scan inside `discoverAndLoadCustomTools()` is + * skipped — subagents inherit the parent's discovery result. MCP/image/tts/ + * web-search tools are still resolved per-session because they depend on the + * session's own model and tool selection. + */ + preloadedCustomTools?: LoadedCustomTool[]; /** Shared event bus for tool/extension communication. Default: creates new bus. */ eventBus?: EventBus; @@ -1193,23 +1206,26 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } // Discover rules and bucket them in one pass to avoid repeated scans over large rule sets. - const { ttsrManager, rulebookRules, alwaysApplyRules } = await logger.time("discoverTtsrRules", async () => { - const { TtsrManager } = await import("./export/ttsr"); - const ttsrSettings = settings.getGroup("ttsr"); - const ttsrManager = new TtsrManager(ttsrSettings); - const rulesResult = - options.rules !== undefined - ? { items: options.rules, warnings: undefined } - : await loadCapability(ruleCapability.id, { cwd }); - const { rulebookRules, alwaysApplyRules } = bucketRules(rulesResult.items, ttsrManager, { - builtinRules: ttsrSettings.builtinRules, - disabledRules: ttsrSettings.disabledRules, - }); - if (existingSession.injectedTtsrRules.length > 0) { - ttsrManager.restoreInjected(existingSession.injectedTtsrRules); - } - return { ttsrManager, rulebookRules, alwaysApplyRules }; - }); + const { ttsrManager, rulebookRules, alwaysApplyRules, allRules } = await logger.time( + "discoverTtsrRules", + async () => { + const { TtsrManager } = await import("./export/ttsr"); + const ttsrSettings = settings.getGroup("ttsr"); + const ttsrManager = new TtsrManager(ttsrSettings); + const rulesResult = + options.rules !== undefined + ? { items: options.rules, warnings: undefined } + : await loadCapability(ruleCapability.id, { cwd }); + const { rulebookRules, alwaysApplyRules } = bucketRules(rulesResult.items, ttsrManager, { + builtinRules: ttsrSettings.builtinRules, + disabledRules: ttsrSettings.disabledRules, + }); + if (existingSession.injectedTtsrRules.length > 0) { + ttsrManager.restoreInjected(existingSession.injectedTtsrRules); + } + return { ttsrManager, rulebookRules, alwaysApplyRules, allRules: rulesResult.items }; + }, + ); // Resolve contextFiles up-front (it's needed before tool creation). The // workspace tree scan is slow on large repos and we MUST NOT block startup on @@ -1331,6 +1347,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} contextFiles, workspaceTree: resolvedWorkspaceTree, skills, + rules: allRules, eventBus, outputSchema: options.outputSchema, requireYieldTool: options.requireYieldTool, @@ -1514,22 +1531,28 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} customTools.push(...getSearchTools()); } - // Discover and load custom tools from .omp/tools/, .claude/tools/, etc. + // Discover custom tools from `.omp/tools/`, `.claude/tools/`, plugins, etc. + // Subagents reuse the parent's discovery via `preloadedCustomTools` so the + // filesystem scan only runs once per process tree. MCP/image/tts/web-search + // tools above are still resolved per-session because they depend on the + // session's own model and tool selection. const builtInToolNames = builtinTools.map(t => t.name); - const discoveredCustomTools = await logger.time( - "discoverAndLoadCustomTools", - discoverAndLoadCustomTools, - [], - cwd, - builtInToolNames, - action => queueResolveHandler(toolSession, action), - ); - for (const { path, error } of discoveredCustomTools.errors) { - logger.error("Custom tool load failed", { path, error }); - } - if (discoveredCustomTools.tools.length > 0) { - customTools.push(...discoveredCustomTools.tools.map(loaded => loaded.tool)); + const loadedCustomTools: LoadedCustomTool[] = + options.preloadedCustomTools ?? + (await logger.time("discoverAndLoadCustomTools", async () => { + const result = await discoverAndLoadCustomTools([], cwd, builtInToolNames, action => + queueResolveHandler(toolSession, action), + ); + for (const { path, error } of result.errors) { + logger.error("Custom tool load failed", { path, error }); + } + return result.tools; + })); + if (loadedCustomTools.length > 0) { + customTools.push(...loadedCustomTools.map(loaded => loaded.tool)); } + // Forward the discovered tools to subagents so they skip the FS scan. + toolSession.loadedCustomTools = loadedCustomTools; const inlineExtensions: ExtensionFactory[] = options.extensions ? [...options.extensions] : []; inlineExtensions.push((await import("./autoresearch")).createAutoresearchExtension); @@ -1539,12 +1562,22 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Load extensions. A preloaded result (e.g. resolved by the CLI before // session creation so it can classify `@file` args extension-aware without - // a session/breadcrumb existing yet) is reused as-is; otherwise discover now + // a session/breadcrumb existing yet, or threaded from a parent session into + // a subagent) is reused without redoing discovery; otherwise discover now // through the shared helper. Preloaded wins over `disableExtensionDiscovery` - // because the preloaded result already reflects that choice — re-running the - // loader here would double-load. - const extensionsResult: LoadExtensionsResult = - options.preloadedExtensions ?? (await loadSessionExtensions(options, cwd, settings, eventBus)); + // because the preloaded result already reflects that choice — re-running + // the loader here would double-load. + // + // Shallow-clone `extensions` so the inline-extensions push below (and any + // other per-session augmentation) never mutates the caller's array; the + // `runtime` is intentionally shared so flag values set pre-creation flow + // into the live session. + const extensionsResult: LoadExtensionsResult = options.preloadedExtensions + ? { ...options.preloadedExtensions, extensions: [...options.preloadedExtensions.extensions] } + : await loadSessionExtensions(options, cwd, settings, eventBus); + // Forward the resolved extensions result to subagents so they reuse it + // instead of repeating `loadSessionExtensions` discovery. + toolSession.extensionsResult = extensionsResult; // Load inline extensions from factories if (inlineExtensions.length > 0) { diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 94d3f7a17..6c3b3c9c6 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -8,14 +8,16 @@ import path from "node:path"; import type { AgentEvent, AgentIdentity, AgentTelemetryConfig, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { recordHandoff, resolveTelemetry } from "@oh-my-pi/pi-agent-core"; import { logger, prompt, untilAborted } from "@oh-my-pi/pi-utils"; +import type { Rule } from "../capability/rule"; import { ModelRegistry } from "../config/model-registry"; import { resolveModelOverrideWithAuthFallback } from "../config/model-resolver"; import type { PromptTemplate } from "../config/prompt-templates"; import { Settings } from "../config/settings"; import { SETTINGS_SCHEMA, type SettingPath } from "../config/settings-schema"; -import type { CustomTool } from "../extensibility/custom-tools/types"; +import type { CustomTool, LoadedCustomTool } from "../extensibility/custom-tools/types"; import { runExtensionCompact, runExtensionSetModel } from "../extensibility/extensions/compact-handler"; import { getSessionSlashCommands } from "../extensibility/extensions/get-commands-handler"; +import type { LoadExtensionsResult } from "../extensibility/extensions/types"; import { buildSkillPromptMessage, type Skill } from "../extensibility/skills"; import type { HindsightSessionState } from "../hindsight/state"; import type { LocalProtocolOptions } from "../internal-urls"; @@ -190,6 +192,12 @@ export interface ExecutorOptions { skills?: Skill[]; promptTemplates?: PromptTemplate[]; workspaceTree?: WorkspaceTree; + /** Parent-discovered rules, forwarded to skip rule discovery in the subagent. */ + rules?: Rule[]; + /** Parent-loaded extensions, forwarded via `preloadedExtensions` to skip discovery. */ + preloadedExtensions?: LoadExtensionsResult; + /** Parent-discovered custom tools, forwarded to skip the `.omp/tools/` scan. */ + preloadedCustomTools?: LoadedCustomTool[]; mcpManager?: MCPManager; authStorage?: AuthStorage; modelRegistry?: ModelRegistry; @@ -1284,6 +1292,9 @@ export async function runSubprocess(options: ExecutorOptions): Promise { const subagentPrompt = prompt.render(subagentSystemPromptTemplate, { agent: agent.systemPrompt, diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index a6fdf4d6b..6d3157b25 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -990,6 +990,9 @@ export class TaskTool implements AgentTool { + let sharedDir: string; + let authStorage: AuthStorage; + let modelRegistry: ModelRegistry; + + beforeAll(async () => { + sharedDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-preloaded-ext-")); + authStorage = await AuthStorage.create(path.join(sharedDir, "auth.db")); + modelRegistry = new ModelRegistry(authStorage, path.join(sharedDir, "models.yml")); + }); + + afterAll(() => { + authStorage.close(); + fs.rmSync(sharedDir, { recursive: true, force: true }); + }); + + it("does not mutate the caller's extensions array when preloadedExtensions is provided", async () => { + const preloaded: LoadExtensionsResult = { + extensions: [], + errors: [], + runtime: { + flagValues: new Map(), + pendingProviderRegistrations: [], + // Cast: only the fields we touch matter; the SDK happily accepts a + // minimal runtime when no extension hooks fire. + } as unknown as LoadExtensionsResult["runtime"], + }; + const beforeLength = preloaded.extensions.length; + const beforeArrayRef = preloaded.extensions; + + await createAgentSession({ + cwd: sharedDir, + agentDir: sharedDir, + sessionManager: SessionManager.inMemory(), + modelRegistry, + settings: Settings.isolated(), + preloadedExtensions: preloaded, + // Disable everything that would touch the network / FS scans. + enableLsp: false, + enableMCP: false, + skipPythonPreflight: true, + skills: [], + rules: [], + preloadedCustomTools: [], + contextFiles: [], + promptTemplates: [], + }); + + // The session's own `extensionsResult` carries inline wrappers, but the + // caller's array (and its identity) must be untouched. + expect(preloaded.extensions).toBe(beforeArrayRef); + expect(preloaded.extensions.length).toBe(beforeLength); + }); +}); diff --git a/packages/coding-agent/test/task/executor-pass-through.test.ts b/packages/coding-agent/test/task/executor-pass-through.test.ts new file mode 100644 index 000000000..629bf0830 --- /dev/null +++ b/packages/coding-agent/test/task/executor-pass-through.test.ts @@ -0,0 +1,139 @@ +/** + * Verifies parent-discovered rules, extensions, and custom tools are forwarded + * to `createAgentSession` so subagents skip the FS scans the parent already + * paid for. Regression guard for issue #2190. + */ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; +import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { LoadedCustomTool } from "@oh-my-pi/pi-coding-agent/extensibility/custom-tools/types"; +import type { LoadExtensionsResult } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; +import type { CreateAgentSessionResult } from "@oh-my-pi/pi-coding-agent/sdk"; +import * as sdkModule from "@oh-my-pi/pi-coding-agent/sdk"; +import type { AgentSession, AgentSessionEvent, PromptOptions } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { runSubprocess } from "@oh-my-pi/pi-coding-agent/task/executor"; +import type { AgentDefinition } from "@oh-my-pi/pi-coding-agent/task/types"; +import { EventBus } from "@oh-my-pi/pi-coding-agent/utils/event-bus"; + +function createMockSession(onPrompt: (params: { emit: (event: AgentSessionEvent) => void }) => void): AgentSession { + const listeners: Array<(event: AgentSessionEvent) => void> = []; + const emit = (event: AgentSessionEvent) => { + for (const listener of listeners) listener(event); + }; + const session = { + state: { messages: [] }, + agent: { state: { systemPrompt: ["test"] } }, + model: undefined, + extensionRunner: undefined, + sessionManager: { appendSessionInit: () => {} }, + getActiveToolNames: () => ["read", "yield"], + setActiveToolsByName: async (_toolNames: string[]) => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + listeners.push(listener); + return () => { + const index = listeners.indexOf(listener); + if (index >= 0) listeners.splice(index, 1); + }; + }, + prompt: async (_text: string, _options?: PromptOptions) => { + onPrompt({ emit }); + }, + waitForIdle: async () => {}, + getLastAssistantMessage: () => undefined, + abort: async () => {}, + dispose: async () => {}, + }; + return session as unknown as AgentSession; +} + +function yieldEmittingSession(): AgentSession { + return createMockSession(({ emit }) => { + emit({ + type: "tool_execution_end", + toolCallId: "tool-pass-through", + toolName: "yield", + result: { + content: [{ type: "text", text: "Result submitted." }], + details: { status: "success", data: { ok: true } }, + }, + isError: false, + }); + }); +} + +function createSessionResult(session: AgentSession): CreateAgentSessionResult { + return { + session, + extensionsResult: { extensions: [], errors: [], runtime: {} as unknown } as unknown as LoadExtensionsResult, + setToolUIContext: () => {}, + eventBus: new EventBus(), + }; +} + +const baseAgent: AgentDefinition = { + name: "task", + description: "test", + systemPrompt: "test", + source: "bundled", +}; + +const baseOptions = { + cwd: "/tmp", + agent: baseAgent, + task: "do work", + index: 0, + id: "subagent-pass-through", + settings: Settings.isolated(), + modelRegistry: { refresh: async () => {} } as unknown as ModelRegistry, + enableLsp: false, +}; + +describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("forwards rules, preloadedExtensions, and preloadedCustomTools to createAgentSession", async () => { + const session = yieldEmittingSession(); + const spy = vi.spyOn(sdkModule, "createAgentSession").mockResolvedValue(createSessionResult(session)); + + const rules: Rule[] = [{ name: "rule-a" } as unknown as Rule]; + const preloadedExtensions = { + extensions: [], + errors: [], + runtime: { sentinel: "runtime" } as unknown, + } as unknown as LoadExtensionsResult; + const preloadedCustomTools: LoadedCustomTool[] = [ + { path: "tools/x.ts", resolvedPath: "/abs/tools/x.ts", tool: { name: "x" } } as unknown as LoadedCustomTool, + ]; + + const result = await runSubprocess({ + ...baseOptions, + rules, + preloadedExtensions, + preloadedCustomTools, + }); + + expect(result.exitCode).toBe(0); + expect(spy).toHaveBeenCalledTimes(1); + const forwarded = spy.mock.calls[0]?.[0]; + // Identity, not equality: passing a clone would defeat the perf fix. + expect(forwarded?.rules).toBe(rules); + expect(forwarded?.preloadedExtensions).toBe(preloadedExtensions); + expect(forwarded?.preloadedCustomTools).toBe(preloadedCustomTools); + }); + + it("forwards undefined when the parent has not pre-discovered state", async () => { + const session = yieldEmittingSession(); + const spy = vi.spyOn(sdkModule, "createAgentSession").mockResolvedValue(createSessionResult(session)); + + const result = await runSubprocess({ ...baseOptions }); + + expect(result.exitCode).toBe(0); + const forwarded = spy.mock.calls[0]?.[0]; + expect(forwarded?.rules).toBeUndefined(); + expect(forwarded?.preloadedExtensions).toBeUndefined(); + expect(forwarded?.preloadedCustomTools).toBeUndefined(); + }); +}); From dde53a8f18d66c2b2aa229a19d2c5e3a4a002f8e Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 14:09:37 +0000 Subject: [PATCH 17/29] fix(task): forward custom-tool paths to subagents, rebind tools per session MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reviewer flagged that forwarding `LoadedCustomTool[]` from a parent session to a subagent reused tool instances whose factories had closed over the parent's `CustomToolAPI` — `cwd`, `exec`, `pushPendingAction`, and `ui` all pointed at the parent. In isolated tasks the tool would `exec` against the parent worktree and queue pending actions on the parent session. Forward only the path list; let each session rebuild tools through `loadCustomTools` so factories see the right `CustomToolAPI`. - `extensibility/custom-tools/loader.ts`: extract `discoverCustomToolPaths` (FS scan only) from `discoverAndLoadCustomTools`; export `ToolPathWithSource`. The combined helper is now `discoverCustomToolPaths` + `loadCustomTools`. - `sdk.ts`: replace `preloadedCustomTools` (`LoadedCustomTool[]`) with `preloadedCustomToolPaths` (`ToolPathWithSource[]`). The custom-tools block runs `loadCustomTools` unconditionally; only the path scan is skipped when the caller pre-discovered it. - `tools/index.ts`: `ToolSession.loadedCustomTools` → `ToolSession.customToolPaths` for the same reason. - `task/executor.ts` and `task/index.ts`: forward `customToolPaths`. Drop the forward for isolated subagents — the worktree shifts `cwd`, so the subagent re-discovers tools against its own working tree. - New `test/sdk-custom-tools-per-session-binding.test.ts` pins the contract: two `loadCustomTools` calls on the same path with different `cwd` and different `pushPendingAction` callbacks yield distinct tool instances whose factories see the per-call bindings. - Updated `executor-pass-through` and `sdk-preloaded-extensions-isolation` tests for the new option name and added a `ToolPathWithSource` fixture. Refs PR review on #2193 --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/extensibility/custom-tools/loader.ts | 62 +++++++---- packages/coding-agent/src/sdk.ts | 64 +++++------ packages/coding-agent/src/task/executor.ts | 13 ++- packages/coding-agent/src/task/index.ts | 3 +- packages/coding-agent/src/tools/index.ts | 10 +- ...k-custom-tools-per-session-binding.test.ts | 102 ++++++++++++++++++ ...sdk-preloaded-extensions-isolation.test.ts | 2 +- .../test/task/executor-pass-through.test.ts | 14 +-- 9 files changed, 203 insertions(+), 69 deletions(-) create mode 100644 packages/coding-agent/test/sdk-custom-tools-per-session-binding.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c9159d46e..5709b68d0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `task`-spawned subagents repeating filesystem scans the parent had already completed. `ExecutorOptions` and the `createAgentSession()` call inside `runSubprocess()` did not forward `rules`, `preloadedExtensions`, or the discovered `.omp/tools/` (a.k.a. `LoadedCustomTool[]`) results, so each subagent re-ran `loadCapability()`, `loadSessionExtensions()`, and `discoverAndLoadCustomTools()`. The toolsession now caches each (`session.rules`, `session.extensionsResult`, `session.loadedCustomTools`), `runSubprocess()` threads them through, and `createAgentSession()` accepts a new `preloadedCustomTools` option. `preloadedExtensions` is now shallow-cloned inside the SDK so inline-extension augmentation (autoresearch + custom-tools wrapper) can never bleed back into the caller's array ([#2190](https://github.com/can1357/oh-my-pi/issues/2190)). +- Fixed `task`-spawned subagents repeating filesystem scans the parent had already completed. `ExecutorOptions` and the `createAgentSession()` call inside `runSubprocess()` did not forward `rules`, `preloadedExtensions`, or the discovered `.omp/tools/` paths, so each subagent re-ran `loadCapability()`, `loadSessionExtensions()`, and the full `.omp/tools/` walk. The toolsession now caches `session.rules`, `session.extensionsResult`, and `session.customToolPaths`; `runSubprocess()` threads them through; and `createAgentSession()` accepts a new `preloadedCustomToolPaths` option backed by a new exported `discoverCustomToolPaths()` helper. Custom-tool factories are still re-bound per session so each `CustomToolAPI` (cwd, exec, pushPendingAction, UI) targets the right session — only the FS scan is skipped. `preloadedExtensions` is shallow-cloned inside the SDK so inline-extension augmentation (autoresearch + custom-tools wrapper) cannot bleed back into the caller's array ([#2190](https://github.com/can1357/oh-my-pi/issues/2190)). ## [15.10.8] - 2026-06-09 diff --git a/packages/coding-agent/src/extensibility/custom-tools/loader.ts b/packages/coding-agent/src/extensibility/custom-tools/loader.ts index a25c9a0f2..8f30fd6bd 100644 --- a/packages/coding-agent/src/extensibility/custom-tools/loader.ts +++ b/packages/coding-agent/src/extensibility/custom-tools/loader.ts @@ -66,8 +66,10 @@ async function loadTool( } } -/** Tool path with optional source metadata */ -interface ToolPathWithSource { +/** Tool path with optional source metadata, suitable for forwarding from a + * parent session to a subagent so the subagent can re-bind tools to its own + * `CustomToolAPI` without redoing the filesystem scan. */ +export interface ToolPathWithSource { path: string; source?: { provider: string; providerName: string; level: "user" | "project" }; } @@ -189,26 +191,19 @@ export async function loadCustomTools( } /** - * Discover and load tools from standard locations via capability system: - * 1. User and project tools discovered by capability providers - * 2. Installed plugins (~/.omp/plugins/node_modules/*) - * 3. Explicitly configured paths from settings or CLI + * Collect the absolute tool-source paths to load, without importing or + * binding factories. Hot path on session startup — the scan walks + * `.omp/tools/`, `.claude/tools/`, the plugin tree, and any configured paths. + * + * Subagents reuse the parent's collected paths via the SDK's + * `preloadedCustomToolPaths` option, then call `loadCustomTools` themselves + * so each session re-binds factories with its own session-scoped + * `CustomToolAPI` (cwd, exec, pushPendingAction, UI). * * @param configuredPaths - Explicit paths from settings.json and CLI --tool flags * @param cwd - Current working directory - * @param builtInToolNames - Names of built-in tools to check for conflicts */ -export async function discoverAndLoadCustomTools( - configuredPaths: string[], - cwd: string, - builtInToolNames: string[], - pushPendingAction?: (action: { - label: string; - sourceToolName: string; - apply(reason: string): Promise>; - reject?(reason: string): Promise | undefined>; - }) => void, -) { +export async function discoverCustomToolPaths(configuredPaths: string[], cwd: string): Promise { const allPathsWithSources: ToolPathWithSource[] = []; const seen = new Set(); @@ -241,5 +236,34 @@ export async function discoverAndLoadCustomTools( addPath(resolvePath(configPath, cwd), { provider: "config", providerName: "Config", level: "project" }); } - return loadCustomTools(allPathsWithSources, cwd, builtInToolNames, pushPendingAction); + return allPathsWithSources; +} + +/** + * Discover and load tools from standard locations via capability system: + * 1. User and project tools discovered by capability providers + * 2. Installed plugins (~/.omp/plugins/node_modules/*) + * 3. Explicitly configured paths from settings or CLI + * + * Composed of {@link discoverCustomToolPaths} (FS scan) + {@link loadCustomTools} + * (per-session binding). Subagents skip the first step and just call + * `loadCustomTools` against the parent's collected paths. + * + * @param configuredPaths - Explicit paths from settings.json and CLI --tool flags + * @param cwd - Current working directory + * @param builtInToolNames - Names of built-in tools to check for conflicts + */ +export async function discoverAndLoadCustomTools( + configuredPaths: string[], + cwd: string, + builtInToolNames: string[], + pushPendingAction?: (action: { + label: string; + sourceToolName: string; + apply(reason: string): Promise>; + reject?(reason: string): Promise | undefined>; + }) => void, +) { + const pathsWithSources = await discoverCustomToolPaths(configuredPaths, cwd); + return loadCustomTools(pathsWithSources, cwd, builtInToolNames, pushPendingAction); } diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index c5db5814a..95b4cf3b2 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -62,13 +62,8 @@ import { type LoadedCustomCommand, loadCustomCommands as loadCustomCommandsInternal, } from "./extensibility/custom-commands"; -import { discoverAndLoadCustomTools } from "./extensibility/custom-tools"; -import type { - CustomTool, - CustomToolContext, - CustomToolSessionEvent, - LoadedCustomTool, -} from "./extensibility/custom-tools/types"; +import { discoverCustomToolPaths, loadCustomTools, type ToolPathWithSource } from "./extensibility/custom-tools"; +import type { CustomTool, CustomToolContext, CustomToolSessionEvent } from "./extensibility/custom-tools/types"; import { discoverAndLoadExtensions, type ExtensionContext, @@ -347,13 +342,17 @@ export interface CreateAgentSessionOptions { */ preloadedExtensions?: LoadExtensionsResult; /** - * Pre-loaded custom tools from `.omp/tools/`, `.claude/tools/`, plugins, etc. - * When provided, the filesystem-scan inside `discoverAndLoadCustomTools()` is - * skipped — subagents inherit the parent's discovery result. MCP/image/tts/ - * web-search tools are still resolved per-session because they depend on the - * session's own model and tool selection. + * Pre-discovered custom-tool source paths from `.omp/tools/`, `.claude/tools/`, + * plugins, etc. When provided, the filesystem-scan inside + * `discoverCustomToolPaths()` is skipped — subagents inherit the parent's + * scan result and call `loadCustomTools()` themselves so each session binds + * tools to its OWN `CustomToolAPI` (cwd, exec, pushPendingAction, UI). + * + * Forwarding the loaded `LoadedCustomTool[]` instances directly would reuse + * the parent's session-bound API and route tool execution back through the + * parent — wrong for isolated tasks and for pending-action routing. */ - preloadedCustomTools?: LoadedCustomTool[]; + preloadedCustomToolPaths?: ToolPathWithSource[]; /** Shared event bus for tool/extension communication. Default: creates new bus. */ eventBus?: EventBus; @@ -1532,27 +1531,28 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } // Discover custom tools from `.omp/tools/`, `.claude/tools/`, plugins, etc. - // Subagents reuse the parent's discovery via `preloadedCustomTools` so the - // filesystem scan only runs once per process tree. MCP/image/tts/web-search - // tools above are still resolved per-session because they depend on the - // session's own model and tool selection. + // Subagents reuse the parent's scan via `preloadedCustomToolPaths` to skip + // the FS walk, but ALWAYS re-call `loadCustomTools` here so factories bind + // to THIS session's `CustomToolAPI` (cwd, exec, pushPendingAction, UI). + // Forwarding the parent's `LoadedCustomTool[]` directly would route tool + // execution back through the parent — wrong for isolated tasks and for + // pending-action queueing. const builtInToolNames = builtinTools.map(t => t.name); - const loadedCustomTools: LoadedCustomTool[] = - options.preloadedCustomTools ?? - (await logger.time("discoverAndLoadCustomTools", async () => { - const result = await discoverAndLoadCustomTools([], cwd, builtInToolNames, action => - queueResolveHandler(toolSession, action), - ); - for (const { path, error } of result.errors) { - logger.error("Custom tool load failed", { path, error }); - } - return result.tools; - })); - if (loadedCustomTools.length > 0) { - customTools.push(...loadedCustomTools.map(loaded => loaded.tool)); + const customToolPaths: ToolPathWithSource[] = + options.preloadedCustomToolPaths ?? + (await logger.time("discoverCustomToolPaths", () => discoverCustomToolPaths([], cwd))); + const customToolsLoadResult = await logger.time("loadCustomTools", () => + loadCustomTools(customToolPaths, cwd, builtInToolNames, action => queueResolveHandler(toolSession, action)), + ); + for (const { path, error } of customToolsLoadResult.errors) { + logger.error("Custom tool load failed", { path, error }); } - // Forward the discovered tools to subagents so they skip the FS scan. - toolSession.loadedCustomTools = loadedCustomTools; + if (customToolsLoadResult.tools.length > 0) { + customTools.push(...customToolsLoadResult.tools.map(loaded => loaded.tool)); + } + // Forward the path list (NOT the loaded tools) to subagents so they + // re-bind under their own `CustomToolAPI` while skipping the FS scan. + toolSession.customToolPaths = customToolPaths; const inlineExtensions: ExtensionFactory[] = options.extensions ? [...options.extensions] : []; inlineExtensions.push((await import("./autoresearch")).createAutoresearchExtension); diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 6c3b3c9c6..a95a92a25 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -14,7 +14,8 @@ import { resolveModelOverrideWithAuthFallback } from "../config/model-resolver"; import type { PromptTemplate } from "../config/prompt-templates"; import { Settings } from "../config/settings"; import { SETTINGS_SCHEMA, type SettingPath } from "../config/settings-schema"; -import type { CustomTool, LoadedCustomTool } from "../extensibility/custom-tools/types"; +import type { ToolPathWithSource } from "../extensibility/custom-tools"; +import type { CustomTool } from "../extensibility/custom-tools/types"; import { runExtensionCompact, runExtensionSetModel } from "../extensibility/extensions/compact-handler"; import { getSessionSlashCommands } from "../extensibility/extensions/get-commands-handler"; import type { LoadExtensionsResult } from "../extensibility/extensions/types"; @@ -196,8 +197,12 @@ export interface ExecutorOptions { rules?: Rule[]; /** Parent-loaded extensions, forwarded via `preloadedExtensions` to skip discovery. */ preloadedExtensions?: LoadExtensionsResult; - /** Parent-discovered custom tools, forwarded to skip the `.omp/tools/` scan. */ - preloadedCustomTools?: LoadedCustomTool[]; + /** + * Parent's discovered custom-tool source paths. Forwarded to skip the + * `.omp/tools/` FS scan in the subagent; the subagent then re-binds each + * tool against its own `CustomToolAPI` (cwd, exec, pushPendingAction, UI). + */ + preloadedCustomToolPaths?: ToolPathWithSource[]; mcpManager?: MCPManager; authStorage?: AuthStorage; modelRegistry?: ModelRegistry; @@ -1294,7 +1299,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise { const subagentPrompt = prompt.render(subagentSystemPromptTemplate, { agent: agent.systemPrompt, diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 6d3157b25..8a2464edb 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -992,7 +992,7 @@ export class TaskTool implements AgentTool { + let tmp: string; + let toolPath: string; + + beforeAll(async () => { + tmp = await fs.mkdtemp(path.join(os.tmpdir(), "pi-custom-tool-binding-")); + toolPath = path.join(tmp, "echo-cwd.ts"); + // Factory exposes the API it was bound to so the test can inspect it. + await fs.writeFile( + toolPath, + [ + "export default function (api) {", + " return {", + " name: 'echo_cwd_' + api.cwd.replace(/[^a-z0-9]/gi, '_'),", + " description: 'returns the cwd the factory was bound to',", + " params: api.typebox.Type.Object({}),", + " async execute() { return { content: [{ type: 'text', text: api.cwd }] }; },", + " __boundApi: api,", + " };", + "}", + ].join("\n"), + ); + }); + + afterAll(async () => { + await fs.rm(tmp, { recursive: true, force: true }); + }); + + it("binds each load to the cwd passed to loadCustomTools", async () => { + const paths: ToolPathWithSource[] = [{ path: toolPath }]; + const parentResult = await loadCustomTools(paths, "/tmp/parent-cwd", []); + const subagentResult = await loadCustomTools(paths, "/tmp/subagent-cwd", []); + + expect(parentResult.errors).toEqual([]); + expect(subagentResult.errors).toEqual([]); + expect(parentResult.tools).toHaveLength(1); + expect(subagentResult.tools).toHaveLength(1); + + const parentApi = (parentResult.tools[0]?.tool as unknown as { __boundApi: CustomToolAPI }).__boundApi; + const subagentApi = (subagentResult.tools[0]?.tool as unknown as { __boundApi: CustomToolAPI }).__boundApi; + + expect(parentApi.cwd).toBe("/tmp/parent-cwd"); + expect(subagentApi.cwd).toBe("/tmp/subagent-cwd"); + expect(subagentApi).not.toBe(parentApi); + // Different tool instances — a session must never see the other's tool. + expect(subagentResult.tools[0]?.tool).not.toBe(parentResult.tools[0]?.tool); + }); + + it("routes pushPendingAction to the loader's own callback, not a shared one", async () => { + const parentLog: string[] = []; + const subagentLog: string[] = []; + + const parentResult = await loadCustomTools([{ path: toolPath }], "/tmp/parent-cwd", [], action => + parentLog.push(`parent:${action.label}`), + ); + const subagentResult = await loadCustomTools([{ path: toolPath }], "/tmp/subagent-cwd", [], action => + subagentLog.push(`subagent:${action.label}`), + ); + + const parentApi = (parentResult.tools[0]?.tool as unknown as { __boundApi: CustomToolAPI }).__boundApi; + const subagentApi = (subagentResult.tools[0]?.tool as unknown as { __boundApi: CustomToolAPI }).__boundApi; + + // Cast: the test fixture exposes the runtime API verbatim. + parentApi.pushPendingAction({ + label: "ping", + sourceToolName: "echo", + apply: async () => ({ content: [] }), + }); + subagentApi.pushPendingAction({ + label: "ping", + sourceToolName: "echo", + apply: async () => ({ content: [] }), + }); + + expect(parentLog).toEqual(["parent:ping"]); + expect(subagentLog).toEqual(["subagent:ping"]); + }); +}); diff --git a/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts b/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts index dbe533a1b..f940c56ce 100644 --- a/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts +++ b/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts @@ -62,7 +62,7 @@ describe("createAgentSession preloadedExtensions isolation (issue #2190)", () => skipPythonPreflight: true, skills: [], rules: [], - preloadedCustomTools: [], + preloadedCustomToolPaths: [], contextFiles: [], promptTemplates: [], }); diff --git a/packages/coding-agent/test/task/executor-pass-through.test.ts b/packages/coding-agent/test/task/executor-pass-through.test.ts index 629bf0830..a516775ed 100644 --- a/packages/coding-agent/test/task/executor-pass-through.test.ts +++ b/packages/coding-agent/test/task/executor-pass-through.test.ts @@ -7,7 +7,7 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import type { LoadedCustomTool } from "@oh-my-pi/pi-coding-agent/extensibility/custom-tools/types"; +import type { ToolPathWithSource } from "@oh-my-pi/pi-coding-agent/extensibility/custom-tools"; import type { LoadExtensionsResult } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; import type { CreateAgentSessionResult } from "@oh-my-pi/pi-coding-agent/sdk"; import * as sdkModule from "@oh-my-pi/pi-coding-agent/sdk"; @@ -94,7 +94,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { vi.restoreAllMocks(); }); - it("forwards rules, preloadedExtensions, and preloadedCustomTools to createAgentSession", async () => { + it("forwards rules, preloadedExtensions, and preloadedCustomToolPaths to createAgentSession", async () => { const session = yieldEmittingSession(); const spy = vi.spyOn(sdkModule, "createAgentSession").mockResolvedValue(createSessionResult(session)); @@ -104,15 +104,15 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { errors: [], runtime: { sentinel: "runtime" } as unknown, } as unknown as LoadExtensionsResult; - const preloadedCustomTools: LoadedCustomTool[] = [ - { path: "tools/x.ts", resolvedPath: "/abs/tools/x.ts", tool: { name: "x" } } as unknown as LoadedCustomTool, + const preloadedCustomToolPaths: ToolPathWithSource[] = [ + { path: "tools/x.ts", source: { provider: "config", providerName: "Config", level: "project" } }, ]; const result = await runSubprocess({ ...baseOptions, rules, preloadedExtensions, - preloadedCustomTools, + preloadedCustomToolPaths, }); expect(result.exitCode).toBe(0); @@ -121,7 +121,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { // Identity, not equality: passing a clone would defeat the perf fix. expect(forwarded?.rules).toBe(rules); expect(forwarded?.preloadedExtensions).toBe(preloadedExtensions); - expect(forwarded?.preloadedCustomTools).toBe(preloadedCustomTools); + expect(forwarded?.preloadedCustomToolPaths).toBe(preloadedCustomToolPaths); }); it("forwards undefined when the parent has not pre-discovered state", async () => { @@ -134,6 +134,6 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const forwarded = spy.mock.calls[0]?.[0]; expect(forwarded?.rules).toBeUndefined(); expect(forwarded?.preloadedExtensions).toBeUndefined(); - expect(forwarded?.preloadedCustomTools).toBeUndefined(); + expect(forwarded?.preloadedCustomToolPaths).toBeUndefined(); }); }); From fe34d0f3925ae054aefd0fc51a7cd079f915131c Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 14:19:47 +0000 Subject: [PATCH 18/29] fix(task): forward extension paths to subagents, rebind extensions per session MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same shape of bug the reviewer flagged for custom tools: forwarding `LoadExtensionsResult` from parent to subagent reused Extension instances whose factories closed over the parent's `ExtensionAPI` — cwd, eventBus, and runtime all pointed at the parent. Any tool/handler/command that referenced `api.exec()`, `api.events`, or `api.runtime` still acted on the parent session/worktree from inside an isolated subagent. Forward only the path list; each session rebuilds extensions through `loadExtensions` so factories see the right `ExtensionAPI`. - `extensibility/extensions/loader.ts`: extract `discoverExtensionPaths` (FS scan only) from `discoverAndLoadExtensions`. The combined helper now composes the two. New export added to the package barrel. - `sdk.ts`: - Add `discoverSessionExtensionPaths()` (the `disableExtensionDiscovery`-aware path-only counterpart of `loadSessionExtensions`). - Add `preloadedExtensionPaths?: string[]` to `CreateAgentSessionOptions`. Three loader branches: `preloadedExtensions` (CLI same-process reuse, still shallow-cloned), `preloadedExtensionPaths` (subagent: skip scan, reload locally), or full discovery. - Document `preloadedExtensions` as same-process-only; subagent forwarding MUST use `preloadedExtensionPaths`. - `tools/index.ts`: `ToolSession.extensionsResult` → `extensionPaths: string[]` for the same reason. - `task/executor.ts` and `task/index.ts`: forward `extensionPaths`. Drop the forward for the isolated `runSubprocess` branch — worktree cwd ≠ parent cwd, so the subagent re-discovers extensions against its own tree. - New `test/sdk-extensions-per-session-binding.test.ts` pins the contract: two `loadExtensions` calls on the same path with different `cwd` and different `EventBus` instances yield distinct Extension + runtime objects whose factories close over the per-call bindings. - Updated `executor-pass-through` and `sdk-preloaded-extensions-isolation` tests for the new option name and comment context. Refs PR review on #2193 --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/extensibility/extensions/index.ts | 1 + .../src/extensibility/extensions/loader.ts | 35 ++++- packages/coding-agent/src/sdk.ts | 123 ++++++++++++------ packages/coding-agent/src/task/executor.ts | 11 +- packages/coding-agent/src/task/index.ts | 3 +- packages/coding-agent/src/tools/index.ts | 10 +- ...sdk-extensions-per-session-binding.test.ts | 85 ++++++++++++ ...sdk-preloaded-extensions-isolation.test.ts | 17 ++- .../test/task/executor-pass-through.test.ts | 14 +- 10 files changed, 232 insertions(+), 69 deletions(-) create mode 100644 packages/coding-agent/test/sdk-extensions-per-session-binding.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5709b68d0..a6790abef 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `task`-spawned subagents repeating filesystem scans the parent had already completed. `ExecutorOptions` and the `createAgentSession()` call inside `runSubprocess()` did not forward `rules`, `preloadedExtensions`, or the discovered `.omp/tools/` paths, so each subagent re-ran `loadCapability()`, `loadSessionExtensions()`, and the full `.omp/tools/` walk. The toolsession now caches `session.rules`, `session.extensionsResult`, and `session.customToolPaths`; `runSubprocess()` threads them through; and `createAgentSession()` accepts a new `preloadedCustomToolPaths` option backed by a new exported `discoverCustomToolPaths()` helper. Custom-tool factories are still re-bound per session so each `CustomToolAPI` (cwd, exec, pushPendingAction, UI) targets the right session — only the FS scan is skipped. `preloadedExtensions` is shallow-cloned inside the SDK so inline-extension augmentation (autoresearch + custom-tools wrapper) cannot bleed back into the caller's array ([#2190](https://github.com/can1357/oh-my-pi/issues/2190)). +- Fixed `task`-spawned subagents repeating filesystem scans the parent had already completed. `ExecutorOptions` and the `createAgentSession()` call inside `runSubprocess()` did not forward `rules`, the discovered extension paths, or the discovered `.omp/tools/` paths, so each subagent re-ran `loadCapability()`, `discoverAndLoadExtensions()`, and the full `.omp/tools/` walk. The toolsession now caches `session.rules`, `session.extensionPaths`, and `session.customToolPaths`; `runSubprocess()` threads them through; and `createAgentSession()` accepts new `preloadedExtensionPaths` and `preloadedCustomToolPaths` options backed by new exported `discoverExtensionPaths()` and `discoverCustomToolPaths()` helpers. Crucially, only path lists are forwarded — never loaded instances. Each session rebuilds its own `Extension` and `LoadedCustomTool` objects so the per-session `ExtensionAPI`/`CustomToolAPI` (cwd, eventBus, runtime, exec, pushPendingAction, UI) targets the right session; forwarding loaded instances would have routed extension handlers and custom-tool execution back through the parent. The CLI's `preloadedExtensions` short-circuit is preserved for same-process reuse and now shallow-clones the caller's `extensions` array so inline-extension augmentation (autoresearch + custom-tools wrapper) cannot bleed back into it ([#2190](https://github.com/can1357/oh-my-pi/issues/2190)). ## [15.10.8] - 2026-06-09 diff --git a/packages/coding-agent/src/extensibility/extensions/index.ts b/packages/coding-agent/src/extensibility/extensions/index.ts index 8a06778c7..c28e524bd 100644 --- a/packages/coding-agent/src/extensibility/extensions/index.ts +++ b/packages/coding-agent/src/extensibility/extensions/index.ts @@ -5,6 +5,7 @@ export type { SlashCommandInfo, SlashCommandLocation, SlashCommandSource } from "../slash-commands"; export { discoverAndLoadExtensions, + discoverExtensionPaths, ExtensionRuntimeNotInitializedError, loadExtensionFromFactory, loadExtensions, diff --git a/packages/coding-agent/src/extensibility/extensions/loader.ts b/packages/coding-agent/src/extensibility/extensions/loader.ts index 40f1c4da9..54639e380 100644 --- a/packages/coding-agent/src/extensibility/extensions/loader.ts +++ b/packages/coding-agent/src/extensibility/extensions/loader.ts @@ -475,16 +475,24 @@ async function discoverExtensionsInDir(dir: string): Promise { return discovered; } - /** - * Discover and load extensions from standard locations. + * Discover absolute paths of extensions to load, without importing or + * binding factories. Hot path on session startup — the scan walks native + * `.omp`/`.pi` extension capabilities, the installed-plugin tree, and any + * configured paths. + * + * Subagents reuse the parent's collected paths via the SDK's + * `preloadedExtensionPaths` option, then call {@link loadExtensions} themselves + * so each session rebuilds Extension instances bound to its OWN + * `ExtensionAPI` (cwd, eventBus, runtime). Forwarding the parent's + * `LoadExtensionsResult` directly would reuse handlers/tools/commands that + * closed over the parent's `cwd` and event bus. */ -export async function discoverAndLoadExtensions( +export async function discoverExtensionPaths( configuredPaths: string[], cwd: string, - eventBus?: EventBus, disabledExtensionIds: string[] = [], -): Promise { +): Promise { const allPaths: string[] = []; const seen = new Set(); const disabled = new Set(disabledExtensionIds); @@ -545,5 +553,20 @@ export async function discoverAndLoadExtensions( addPath(resolved); } - return loadExtensions(allPaths, cwd, eventBus); + return allPaths; +} + +/** + * Discover and load extensions from standard locations. Composed of + * {@link discoverExtensionPaths} (FS scan) + {@link loadExtensions} + * (per-session binding). + */ +export async function discoverAndLoadExtensions( + configuredPaths: string[], + cwd: string, + eventBus?: EventBus, + disabledExtensionIds: string[] = [], +): Promise { + const paths = await discoverExtensionPaths(configuredPaths, cwd, disabledExtensionIds); + return loadExtensions(paths, cwd, eventBus); } diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 95b4cf3b2..730c0bca1 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -66,6 +66,7 @@ import { discoverCustomToolPaths, loadCustomTools, type ToolPathWithSource } fro import type { CustomTool, CustomToolContext, CustomToolSessionEvent } from "./extensibility/custom-tools/types"; import { discoverAndLoadExtensions, + discoverExtensionPaths, type ExtensionContext, type ExtensionFactory, ExtensionRunner, @@ -337,10 +338,29 @@ export interface CreateAgentSessionOptions { /** Disable extension discovery (explicit paths still load). */ disableExtensionDiscovery?: boolean; /** - * Pre-loaded extensions (skips file discovery). - * @internal Used by CLI when extensions are loaded early to parse custom flags. + * Pre-loaded extensions (skips file discovery and the per-session factory + * call). Used by the CLI when extensions are loaded early to parse custom + * flags — the same process owns the returned instances, so reusing them is + * safe. + * + * NEVER pass this across session boundaries (e.g. parent → subagent). + * `Extension` instances close over a parent-bound `ExtensionAPI` (cwd, + * eventBus, runtime), and reusing them would route tools/handlers/commands + * back through the parent. For subagents, forward + * {@link preloadedExtensionPaths} instead. + * + * @internal */ preloadedExtensions?: LoadExtensionsResult; + /** + * Pre-discovered extension source paths. When provided, the filesystem-scan + * inside `discoverExtensionPaths()` is skipped — the session still calls + * `loadExtensions()` itself so each `Extension` is bound to THIS session's + * `ExtensionAPI` (cwd, eventBus, runtime). + * + * This is the safe pass-through for parent → subagent forwarding. + */ + preloadedExtensionPaths?: string[]; /** * Pre-discovered custom-tool source paths from `.omp/tools/`, `.claude/tools/`, * plugins, etc. When provided, the filesystem-scan inside @@ -577,6 +597,26 @@ export async function discoverExtensions(cwd?: string): Promise, + cwd: string, + settings: Settings, +): Promise { + if (options.disableExtensionDiscovery) { + return options.additionalExtensionPaths ?? []; + } + const configuredPaths = [...(options.additionalExtensionPaths ?? []), ...(settings.get("extensions") ?? [])]; + const disabledExtensionIds = settings.get("disabledExtensions") ?? []; + return discoverExtensionPaths(configuredPaths, cwd, disabledExtensionIds); +} + /** * Load the discovered/configured extensions for a session — everything {@link * createAgentSession} would load except the inline factory extensions it appends @@ -592,23 +632,8 @@ export async function loadSessionExtensions( settings: Settings, eventBus: EventBus, ): Promise { - let result: LoadExtensionsResult; - if (options.disableExtensionDiscovery) { - const configuredPaths = options.additionalExtensionPaths ?? []; - result = await logger.time("loadExtensions", loadExtensions, configuredPaths, cwd, eventBus); - } else { - // Merge CLI extension paths with settings extension paths. - const configuredPaths = [...(options.additionalExtensionPaths ?? []), ...(settings.get("extensions") ?? [])]; - const disabledExtensionIds = settings.get("disabledExtensions") ?? []; - result = await logger.time( - "discoverAndLoadExtensions", - discoverAndLoadExtensions, - configuredPaths, - cwd, - eventBus, - disabledExtensionIds, - ); - } + const paths = await discoverSessionExtensionPaths(options, cwd, settings); + const result = await logger.time("loadExtensions", loadExtensions, paths, cwd, eventBus); for (const { path, error } of result.errors) { logger.error("Failed to load extension", { path, error }); } @@ -1560,24 +1585,48 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} inlineExtensions.push(createCustomToolsExtension(customTools)); } - // Load extensions. A preloaded result (e.g. resolved by the CLI before - // session creation so it can classify `@file` args extension-aware without - // a session/breadcrumb existing yet, or threaded from a parent session into - // a subagent) is reused without redoing discovery; otherwise discover now - // through the shared helper. Preloaded wins over `disableExtensionDiscovery` - // because the preloaded result already reflects that choice — re-running - // the loader here would double-load. - // - // Shallow-clone `extensions` so the inline-extensions push below (and any - // other per-session augmentation) never mutates the caller's array; the - // `runtime` is intentionally shared so flag values set pre-creation flow - // into the live session. - const extensionsResult: LoadExtensionsResult = options.preloadedExtensions - ? { ...options.preloadedExtensions, extensions: [...options.preloadedExtensions.extensions] } - : await loadSessionExtensions(options, cwd, settings, eventBus); - // Forward the resolved extensions result to subagents so they reuse it - // instead of repeating `loadSessionExtensions` discovery. - toolSession.extensionsResult = extensionsResult; + // Load extensions. Three paths: + // 1. `preloadedExtensions` (CLI): caller already loaded — reuse the + // Extension instances. Shallow-clone `extensions` so the inline + // push below cannot mutate the caller's array. `runtime` is shared + // so flag values set pre-creation flow into the live session. + // 2. `preloadedExtensionPaths` (subagent): caller resolved paths; + // skip the FS scan but always re-call `loadExtensions` here so + // each `Extension` binds to THIS session's `ExtensionAPI` + // (cwd, eventBus, runtime). + // 3. No preload: run the full session discovery. + // `disableExtensionDiscovery` is honored implicitly: a caller that set + // the flag and pre-resolved the result already reflects that choice. + let extensionPaths: string[]; + let extensionsResult: LoadExtensionsResult; + if (options.preloadedExtensions) { + extensionsResult = { + ...options.preloadedExtensions, + extensions: [...options.preloadedExtensions.extensions], + }; + // Capture paths for downstream forwarding; filter inline-factory + // entries (``) — those are per-session, not source paths. + extensionPaths = extensionsResult.extensions + .map(ext => ext.resolvedPath) + .filter(p => !p.startsWith(" + discoverSessionExtensionPaths(options, cwd, settings), + ); + extensionsResult = await logger.time("loadExtensions", loadExtensions, extensionPaths, cwd, eventBus); + for (const { path, error } of extensionsResult.errors) { + logger.error("Failed to load extension", { path, error }); + } + } + // Forward the source-path list (NOT the loaded instances) so subagents + // rebuild their own session-scoped extensions. + toolSession.extensionPaths = extensionPaths; // Load inline extensions from factories if (inlineExtensions.length > 0) { diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index a95a92a25..0337e5579 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -18,7 +18,6 @@ import type { ToolPathWithSource } from "../extensibility/custom-tools"; import type { CustomTool } from "../extensibility/custom-tools/types"; import { runExtensionCompact, runExtensionSetModel } from "../extensibility/extensions/compact-handler"; import { getSessionSlashCommands } from "../extensibility/extensions/get-commands-handler"; -import type { LoadExtensionsResult } from "../extensibility/extensions/types"; import { buildSkillPromptMessage, type Skill } from "../extensibility/skills"; import type { HindsightSessionState } from "../hindsight/state"; import type { LocalProtocolOptions } from "../internal-urls"; @@ -195,8 +194,12 @@ export interface ExecutorOptions { workspaceTree?: WorkspaceTree; /** Parent-discovered rules, forwarded to skip rule discovery in the subagent. */ rules?: Rule[]; - /** Parent-loaded extensions, forwarded via `preloadedExtensions` to skip discovery. */ - preloadedExtensions?: LoadExtensionsResult; + /** + * Parent's discovered extension source paths. Forwarded to skip the + * extension FS scan in the subagent; the subagent then re-binds each + * extension against its own `ExtensionAPI` (cwd, eventBus, runtime). + */ + preloadedExtensionPaths?: string[]; /** * Parent's discovered custom-tool source paths. Forwarded to skip the * `.omp/tools/` FS scan in the subagent; the subagent then re-binds each @@ -1298,7 +1301,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise { const subagentPrompt = prompt.render(subagentSystemPromptTemplate, { diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 8a2464edb..26f2d116c 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -991,7 +991,7 @@ export class TaskTool implements AgentTool`) are NOT included — those are session-local. + */ + extensionPaths?: string[]; /** * Pre-discovered custom-tool source paths from `.omp/tools/`, `.claude/tools/`, * plugins, etc. Forwarded to subagents so they skip the FS scan but still diff --git a/packages/coding-agent/test/sdk-extensions-per-session-binding.test.ts b/packages/coding-agent/test/sdk-extensions-per-session-binding.test.ts new file mode 100644 index 000000000..18c03815c --- /dev/null +++ b/packages/coding-agent/test/sdk-extensions-per-session-binding.test.ts @@ -0,0 +1,85 @@ +/** + * Regression guard for PR review feedback on #2190. + * + * Subagents inherit the parent's extension source *paths* (a cheap FS scan + * the parent already paid for), but each session MUST rebuild its own + * `Extension` instances so factories see the subagent's `ExtensionAPI` + * (cwd, eventBus, runtime). Forwarding the parent's loaded Extension + * instances would have tools/handlers/commands close over the parent's + * `cwd` and event bus — wrong for isolated tasks. + * + * Pins down `loadExtensions()` so the SDK can rely on it returning fresh + * Extension instances per call. + */ +import { afterAll, beforeAll, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { loadExtensions } from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; +import { EventBus } from "@oh-my-pi/pi-coding-agent/utils/event-bus"; + +describe("loadExtensions per-session binding (#2190 review fix)", () => { + let tmp: string; + let extPath: string; + + beforeAll(async () => { + tmp = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ext-binding-")); + extPath = path.join(tmp, "record-cwd.ts"); + // Factory tags the extension with the cwd + events it was bound to so + // the test can inspect what closures captured. + await fs.writeFile( + extPath, + [ + "export default function (api) {", + " api.registerTool({", + " name: 'tag',", + " description: 'binding probe',", + " params: api.typebox.Type.Object({}),", + " async execute() { return { content: [{ type: 'text', text: '' }] }; },", + " });", + " Object.defineProperty(globalThis, '__lastExtBinding', {", + " value: { cwd: api.exec.toString().includes('cwd') ? api : api, events: api.events },", + " writable: true,", + " configurable: true,", + " });", + " globalThis.__bindings = globalThis.__bindings || [];", + " globalThis.__bindings.push({ events: api.events });", + "}", + ].join("\n"), + ); + }); + + afterAll(async () => { + await fs.rm(tmp, { recursive: true, force: true }); + delete (globalThis as { __bindings?: unknown }).__bindings; + delete (globalThis as { __lastExtBinding?: unknown }).__lastExtBinding; + }); + + it("creates a distinct Extension and ExtensionAPI per call (fresh eventBus + runtime)", async () => { + (globalThis as { __bindings?: { events: EventBus }[] }).__bindings = []; + + const parentEventBus = new EventBus(); + const subagentEventBus = new EventBus(); + expect(parentEventBus).not.toBe(subagentEventBus); + + const parent = await loadExtensions([extPath], "/tmp/parent-cwd", parentEventBus); + const subagent = await loadExtensions([extPath], "/tmp/subagent-cwd", subagentEventBus); + + expect(parent.errors).toEqual([]); + expect(subagent.errors).toEqual([]); + expect(parent.extensions).toHaveLength(1); + expect(subagent.extensions).toHaveLength(1); + + // Distinct Extension instances — the subagent must never share with parent. + expect(subagent.extensions[0]).not.toBe(parent.extensions[0]); + // Distinct ExtensionRuntime instances — flagValues and pendingProviderRegistrations + // MUST NOT be shared, or per-session flags/registrations bleed across. + expect(subagent.runtime).not.toBe(parent.runtime); + + // Each factory saw the eventBus passed to its own loadExtensions call. + const bindings = (globalThis as { __bindings?: { events: EventBus }[] }).__bindings ?? []; + expect(bindings).toHaveLength(2); + expect(bindings[0]?.events).toBe(parentEventBus); + expect(bindings[1]?.events).toBe(subagentEventBus); + }); +}); diff --git a/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts b/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts index f940c56ce..e37e757bf 100644 --- a/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts +++ b/packages/coding-agent/test/sdk-preloaded-extensions-isolation.test.ts @@ -1,12 +1,15 @@ /** - * Regression guard for issue #2190. + * Regression guard for issue #2190 / PR #2193 review. * - * Subagents pass the parent session's `extensionsResult` as - * `preloadedExtensions` so the extension-discovery work isn't repeated. - * `createAgentSession()` augments the result with inline extensions - * (autoresearch + custom-tools wrapper), so it MUST clone the caller's - * `extensions` array before mutating it — otherwise the parent's runtime - * accumulates every subagent's inline wrappers. + * The CLI loads extensions early to parse custom flags, then hands the result + * back through `preloadedExtensions` so its OWN session can reuse the loaded + * instances without redoing the FS scan. `createAgentSession()` augments the + * result with inline extensions (autoresearch + custom-tools wrapper), so it + * MUST clone the caller's `extensions` array before mutating it — otherwise + * the caller's array accumulates session-local wrappers it never authored. + * + * Subagent forwarding is a separate path (`preloadedExtensionPaths`) which + * reloads extensions per session so each session's `ExtensionAPI` is its own. */ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; diff --git a/packages/coding-agent/test/task/executor-pass-through.test.ts b/packages/coding-agent/test/task/executor-pass-through.test.ts index a516775ed..d6017dac0 100644 --- a/packages/coding-agent/test/task/executor-pass-through.test.ts +++ b/packages/coding-agent/test/task/executor-pass-through.test.ts @@ -94,16 +94,12 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { vi.restoreAllMocks(); }); - it("forwards rules, preloadedExtensions, and preloadedCustomToolPaths to createAgentSession", async () => { + it("forwards rules, preloadedExtensionPaths, and preloadedCustomToolPaths to createAgentSession", async () => { const session = yieldEmittingSession(); const spy = vi.spyOn(sdkModule, "createAgentSession").mockResolvedValue(createSessionResult(session)); const rules: Rule[] = [{ name: "rule-a" } as unknown as Rule]; - const preloadedExtensions = { - extensions: [], - errors: [], - runtime: { sentinel: "runtime" } as unknown, - } as unknown as LoadExtensionsResult; + const preloadedExtensionPaths = ["/abs/parent/.omp/extensions/foo.ts"]; const preloadedCustomToolPaths: ToolPathWithSource[] = [ { path: "tools/x.ts", source: { provider: "config", providerName: "Config", level: "project" } }, ]; @@ -111,7 +107,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const result = await runSubprocess({ ...baseOptions, rules, - preloadedExtensions, + preloadedExtensionPaths, preloadedCustomToolPaths, }); @@ -120,7 +116,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const forwarded = spy.mock.calls[0]?.[0]; // Identity, not equality: passing a clone would defeat the perf fix. expect(forwarded?.rules).toBe(rules); - expect(forwarded?.preloadedExtensions).toBe(preloadedExtensions); + expect(forwarded?.preloadedExtensionPaths).toBe(preloadedExtensionPaths); expect(forwarded?.preloadedCustomToolPaths).toBe(preloadedCustomToolPaths); }); @@ -133,7 +129,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { expect(result.exitCode).toBe(0); const forwarded = spy.mock.calls[0]?.[0]; expect(forwarded?.rules).toBeUndefined(); - expect(forwarded?.preloadedExtensions).toBeUndefined(); + expect(forwarded?.preloadedExtensionPaths).toBeUndefined(); expect(forwarded?.preloadedCustomToolPaths).toBeUndefined(); }); }); From dccba5297e9395201f4759f7ea722647164ec6ab Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 16:16:03 +0000 Subject: [PATCH 19/29] fix(tui): restored mcp auth hyperlink fallback MCP OAuth fallback prompts now emit an auth-safe terminal hyperlink even when auto-detection disables normal URL hyperlinks, matching the provider login behavior while preserving the raw copy URL.\n\nFixes #2196 --- packages/coding-agent/CHANGELOG.md | 4 +++ .../controllers/mcp-command-controller.ts | 4 +-- packages/coding-agent/src/tui/hyperlink.ts | 27 ++++++++++++++++--- .../mcp-authorization-link.test.ts | 10 +++---- .../coding-agent/test/tui/hyperlink.test.ts | 22 +++++++++++++++ 5 files changed, 56 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bef9ae4ac..a96ce6ac6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed MCP OAuth fallback prompts so the "Click here to authorize" label emits an auth-safe terminal hyperlink even when hyperlink auto-detection is unavailable, keeping non-browser MCP setup usable ([#2196](https://github.com/can1357/oh-my-pi/issues/2196)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts index 377b3e12f..08f933cdd 100644 --- a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts @@ -36,7 +36,7 @@ import { import type { MCPAuthConfig, MCPServerConfig, MCPServerConnection } from "../../mcp/types"; import type { OAuthCredential } from "../../session/auth-storage"; import { shortenPath } from "../../tools/render-utils"; -import { urlHyperlink } from "../../tui"; +import { urlHyperlinkAlways } from "../../tui"; import { openPath } from "../../utils/open"; import { ChatBlock } from "../components/chat-block"; import { MCPAddWizard } from "../components/mcp-add-wizard"; @@ -63,7 +63,7 @@ export class MCPAuthorizationLinkPrompt implements Component { invalidate(): void {} render(_width: number): string[] { - const link = urlHyperlink(this.#url, "Click here to authorize"); + const link = urlHyperlinkAlways(this.#url, "Click here to authorize"); return [ ` ${theme.fg("success", "Open authorization URL:")}`, ` ${theme.fg("accent", link)}`, diff --git a/packages/coding-agent/src/tui/hyperlink.ts b/packages/coding-agent/src/tui/hyperlink.ts index dd9c1fafd..413e30884 100644 --- a/packages/coding-agent/src/tui/hyperlink.ts +++ b/packages/coding-agent/src/tui/hyperlink.ts @@ -18,6 +18,7 @@ import { const OSC = "\x1b]"; const ST = "\x1b\\"; +const BEL = "\x07"; /** Stable 8-char hex ID derived from a URI — hints terminals to coalesce identical adjacent links. */ function buildLinkId(uri: string): string { @@ -60,14 +61,18 @@ function safeHyperlinkUri(uri: string): string | undefined { return uri; } -function wrapHyperlink(uri: string, displayText: string): string { - if (!isHyperlinkEnabled()) return displayText; +function wrapHyperlinkCore(uri: string, displayText: string, terminator: typeof ST | typeof BEL): string { // Do not double-wrap if the text already embeds an OSC 8 sequence. if (displayText.includes("\x1b]8;")) return displayText; const safeUri = safeHyperlinkUri(uri); if (!safeUri) return displayText; const id = buildLinkId(safeUri); - return `${OSC}8;id=${id};${safeUri}${ST}${displayText}${OSC}8;;${ST}`; + return `${OSC}8;id=${id};${safeUri}${terminator}${displayText}${OSC}8;;${terminator}`; +} + +function wrapHyperlink(uri: string, displayText: string): string { + if (!isHyperlinkEnabled()) return displayText; + return wrapHyperlinkCore(uri, displayText, ST); } /** @@ -95,6 +100,22 @@ export function urlHyperlink(url: string, displayText: string): string { } } +/** + * Wrap `displayText` in an OSC 8 hyperlink pointing at an HTTP(S) URL without + * terminal capability gating. Used for auth prompts where an inert "click" + * label blocks login on terminals with incomplete capability detection. + */ +export function urlHyperlinkAlways(url: string, displayText: string): string { + const normalized = url.match(/^www\./i) ? `https://${url}` : url; + try { + const parsed = new URL(normalized); + if (parsed.protocol !== "http:" && parsed.protocol !== "https:") return displayText; + return wrapHyperlinkCore(parsed.href, displayText, BEL); + } catch { + return displayText; + } +} + /** * Wrap `displayText` in an OSC 8 hyperlink pointing at a filesystem path. * diff --git a/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts b/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts index 2fcb3dcca..666ce0b47 100644 --- a/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts +++ b/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts @@ -5,10 +5,10 @@ import { MCPAuthorizationLinkPrompt } from "@oh-my-pi/pi-coding-agent/modes/cont import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; const OSC = "\x1b]"; -const ST = "\x1b\\"; +const BEL = "\x07"; function extractLinkUri(text: string): string | undefined { - return text.match(/\x1b\]8;[^;]*;([^\x1b]+)\x1b\\/)?.[1]; + return text.match(/\x1b\]8;[^;]*;([^\x1b\x07]+)(?:\x1b\\|\x07)/)?.[1]; } const LONG_AUTH_URL = @@ -26,15 +26,13 @@ describe("MCPAuthorizationLinkPrompt", () => { resetSettingsForTest(); }); - it("renders a short clickable label and keeps the copy URL on one line", () => { - settings.override("tui.hyperlinks", "always"); - + it("renders a clickable label even when hyperlink auto-detection is false", () => { const lines = new MCPAuthorizationLinkPrompt(LONG_AUTH_URL).render(80); const plainLines = lines.map(line => stripVTControlCharacters(line)); expect(lines).toHaveLength(3); expect(lines[1]).toContain(`${OSC}8;`); - expect(lines[1]).toContain(`${OSC}8;;${ST}`); + expect(lines[1]).toContain(`${OSC}8;;${BEL}`); expect(extractLinkUri(lines[1])).toBe(LONG_AUTH_URL); expect(plainLines[1]).toContain("Click here to authorize"); expect(plainLines[2]).toBe(` Copy URL: ${LONG_AUTH_URL}`); diff --git a/packages/coding-agent/test/tui/hyperlink.test.ts b/packages/coding-agent/test/tui/hyperlink.test.ts index 733930404..cb56e5e83 100644 --- a/packages/coding-agent/test/tui/hyperlink.test.ts +++ b/packages/coding-agent/test/tui/hyperlink.test.ts @@ -6,12 +6,14 @@ import { tryResolveInternalUrlSync, uriHyperlink, urlHyperlink, + urlHyperlinkAlways, } from "@oh-my-pi/pi-coding-agent/tui/hyperlink"; import * as terminalCaps from "@oh-my-pi/pi-tui"; // OSC 8 sequence markers const OSC = "\x1b]"; const ST = "\x1b\\"; +const BEL = "\x07"; const LINK_END = `${OSC}8;;${ST}`; const ORIGINAL_NO_COLOR = Bun.env.NO_COLOR; @@ -21,6 +23,10 @@ function extractLinkUri(text: string): string | undefined { return match?.[1]; } +function extractAnyTerminatorLinkUri(text: string): string | undefined { + return text.match(/\x1b\]8;[^;]*;([^\x1b\x07]+)(?:\x1b\\|\x07)/)?.[1]; +} + /** Returns true if the string contains an OSC 8 hyperlink wrapping a given display text. */ function isHyperlinked(text: string): boolean { return text.includes(`${OSC}8;`) && text.includes(LINK_END); @@ -217,6 +223,22 @@ describe("urlHyperlink", () => { }); }); +describe("urlHyperlinkAlways", () => { + it("wraps HTTP URLs when hyperlink auto-detection is disabled", () => { + setHyperlinkMode("off"); + const result = urlHyperlinkAlways("www.example.com/path", "example"); + + expect(result).toContain(`${OSC}8;`); + expect(result).toContain(`${OSC}8;;${BEL}`); + expect(extractAnyTerminatorLinkUri(result)).toBe("https://www.example.com/path"); + }); + + it("does not wrap non-HTTP URL schemes", () => { + setHyperlinkMode("off"); + expect(urlHyperlinkAlways("ftp://example.com/file", "file")).toBe("file"); + }); +}); + describe("tryResolveInternalUrlSync", () => { it("returns undefined for non-internal URLs", () => { expect(tryResolveInternalUrlSync("/abs/path/file.ts")).toBeUndefined(); From 9c17f27c047ae4b183c968fc9058593212b48f13 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 16:20:07 +0000 Subject: [PATCH 20/29] fix(tui): honored tui.hyperlinks=off in mcp auth fallback urlHyperlinkAlways now short-circuits to plain text when the user has explicitly opted out via tui.hyperlinks=off, while still bypassing capability auto-detection for auto mode. --- packages/coding-agent/src/tui/hyperlink.ts | 9 ++++++--- packages/coding-agent/test/tui/hyperlink.test.ts | 13 ++++++++++--- 2 files changed, 16 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/src/tui/hyperlink.ts b/packages/coding-agent/src/tui/hyperlink.ts index 413e30884..d5c161469 100644 --- a/packages/coding-agent/src/tui/hyperlink.ts +++ b/packages/coding-agent/src/tui/hyperlink.ts @@ -101,11 +101,14 @@ export function urlHyperlink(url: string, displayText: string): string { } /** - * Wrap `displayText` in an OSC 8 hyperlink pointing at an HTTP(S) URL without - * terminal capability gating. Used for auth prompts where an inert "click" - * label blocks login on terminals with incomplete capability detection. + * Wrap `displayText` in an OSC 8 hyperlink pointing at an HTTP(S) URL, + * bypassing terminal capability auto-detection. Used for auth prompts where + * an inert "click" label blocks login on terminals whose capabilities are + * not advertised. Still returns plain text when the user has explicitly + * opted out via `tui.hyperlinks=off`. */ export function urlHyperlinkAlways(url: string, displayText: string): string { + if (settings.get("tui.hyperlinks") === "off") return displayText; const normalized = url.match(/^www\./i) ? `https://${url}` : url; try { const parsed = new URL(normalized); diff --git a/packages/coding-agent/test/tui/hyperlink.test.ts b/packages/coding-agent/test/tui/hyperlink.test.ts index cb56e5e83..9a5e0742d 100644 --- a/packages/coding-agent/test/tui/hyperlink.test.ts +++ b/packages/coding-agent/test/tui/hyperlink.test.ts @@ -224,17 +224,24 @@ describe("urlHyperlink", () => { }); describe("urlHyperlinkAlways", () => { - it("wraps HTTP URLs when hyperlink auto-detection is disabled", () => { - setHyperlinkMode("off"); + it("wraps HTTP URLs in auto mode even when capability detection would suppress", () => { + setHyperlinkMode("auto"); + Bun.env.NO_COLOR = "1"; // forces isHyperlinkEnabled() to false in auto mode const result = urlHyperlinkAlways("www.example.com/path", "example"); + expect(isHyperlinkEnabled()).toBe(false); expect(result).toContain(`${OSC}8;`); expect(result).toContain(`${OSC}8;;${BEL}`); expect(extractAnyTerminatorLinkUri(result)).toBe("https://www.example.com/path"); }); - it("does not wrap non-HTTP URL schemes", () => { + it("returns plain text when the user opts out with tui.hyperlinks=off", () => { setHyperlinkMode("off"); + expect(urlHyperlinkAlways("https://example.com/path", "example")).toBe("example"); + }); + + it("does not wrap non-HTTP URL schemes", () => { + setHyperlinkMode("always"); expect(urlHyperlinkAlways("ftp://example.com/file", "file")).toBe("file"); }); }); From b63398aeb950ef050afa84676827c773784e99a4 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:45:40 +0200 Subject: [PATCH 21/29] fix(coding-agent): fixed bracketed image pastes to handle multiple paths - Updated bracketed-image parsing to recognize multiple image paths in a single paste, including quoted values and shell-escaped spaces. - Changed the editor image-path handler to process each matched path in order, awaiting async handlers. - Added tests for multi-path routing/normalization and recorded the fix in the coding-agent changelog entry. --- packages/coding-agent/CHANGELOG.md | 2 + .../src/modes/components/custom-editor.ts | 78 ++++++++++++++++--- .../src/modes/controllers/input-controller.ts | 2 +- .../test/custom-editor-keybindings.test.ts | 37 ++++++++- packages/tui/src/components/image.ts | 8 +- packages/tui/src/tui.ts | 46 ++++++++++- 6 files changed, 159 insertions(+), 14 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5f0a87352..b2859b029 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,8 @@ ### Fixed +- Fixed bracketed pastes containing multiple image file paths so each image is attached in order instead of treating the whole paste as one unreadable path. + - Fixed MCP OAuth fallback prompts so the "Click here to authorize" label emits an auth-safe terminal hyperlink even when hyperlink auto-detection is unavailable, keeping non-browser MCP setup usable ([#2196](https://github.com/can1357/oh-my-pi/issues/2196)). ### Fixed diff --git a/packages/coding-agent/src/modes/components/custom-editor.ts b/packages/coding-agent/src/modes/components/custom-editor.ts index 73e42aa46..e6f04e6ff 100644 --- a/packages/coding-agent/src/modes/components/custom-editor.ts +++ b/packages/coding-agent/src/modes/components/custom-editor.ts @@ -58,16 +58,72 @@ function buildMatchKeys(keys: readonly KeyId[]): Set { const BRACKETED_PASTE_START = "\x1b[200~"; const BRACKETED_PASTE_END = "\x1b[201~"; const BRACKETED_IMAGE_PATH_REGEX = /\.(?:png|jpe?g|gif|webp)$/i; +const BRACKETED_IMAGE_PATH_BOUNDARY_REGEX = /\.(?:png|jpe?g|gif|webp)(?=$|["']?\s)/gi; +const SHELL_ESCAPED_PATH_CHAR_REGEX = /\\([\\\s'"()[\]{}&;<>|?*!$`])/g; -export function extractBracketedImagePastePath(data: string): string | undefined { +function isPastedPathSeparator(char: string | undefined): boolean { + return char === undefined || char === " " || char === "\t" || char === "\r" || char === "\n"; +} + +function imagePathBoundaryEnd(payload: string, segmentStart: number, extensionEnd: number): number | undefined { + const quote = payload[segmentStart]; + const afterExtension = payload[extensionEnd]; + if (quote === '"' || quote === "'") { + return afterExtension === quote && isPastedPathSeparator(payload[extensionEnd + 1]) + ? extensionEnd + 1 + : undefined; + } + if (isPastedPathSeparator(afterExtension)) return extensionEnd; + return undefined; +} + +function normalizePastedImagePath(path: string): string { + const trimmed = path.trim(); + const first = trimmed[0]; + const last = trimmed[trimmed.length - 1]; + const unquoted = + trimmed.length > 1 && (first === '"' || first === "'") && last === first ? trimmed.slice(1, -1) : trimmed; + return unquoted.replace(SHELL_ESCAPED_PATH_CHAR_REGEX, "$1"); +} + +export function extractBracketedImagePastePaths(data: string): string[] | undefined { if (!data.startsWith(BRACKETED_PASTE_START)) return undefined; const endIndex = data.indexOf(BRACKETED_PASTE_END, BRACKETED_PASTE_START.length); if (endIndex === -1 || endIndex + BRACKETED_PASTE_END.length !== data.length) return undefined; const pasted = data.slice(BRACKETED_PASTE_START.length, endIndex).trim(); - if (!pasted || /[\r\n]/.test(pasted)) return undefined; - if (!BRACKETED_IMAGE_PATH_REGEX.test(pasted)) return undefined; - return pasted; + if (!pasted) return undefined; + + const paths: string[] = []; + let segmentStart = 0; + BRACKETED_IMAGE_PATH_BOUNDARY_REGEX.lastIndex = 0; + for ( + let match = BRACKETED_IMAGE_PATH_BOUNDARY_REGEX.exec(pasted); + match; + match = BRACKETED_IMAGE_PATH_BOUNDARY_REGEX.exec(pasted) + ) { + const extensionEnd = match.index + match[0].length; + const boundaryEnd = imagePathBoundaryEnd(pasted, segmentStart, extensionEnd); + if (boundaryEnd === undefined) continue; + + const path = normalizePastedImagePath(pasted.slice(segmentStart, boundaryEnd)); + if (!path || !BRACKETED_IMAGE_PATH_REGEX.test(path)) return undefined; + paths.push(path); + + segmentStart = boundaryEnd; + while (segmentStart < pasted.length && isPastedPathSeparator(pasted[segmentStart])) { + segmentStart++; + } + BRACKETED_IMAGE_PATH_BOUNDARY_REGEX.lastIndex = segmentStart; + } + + if (paths.length === 0 || segmentStart !== pasted.length) return undefined; + return paths; +} + +export function extractBracketedImagePastePath(data: string): string | undefined { + const paths = extractBracketedImagePastePaths(data); + return paths?.length === 1 ? paths[0] : undefined; } /** @@ -111,8 +167,8 @@ export class CustomEditor extends Editor { onCopyPrompt?: () => void; /** Called when the configured image-paste shortcut is pressed. */ onPasteImage?: () => Promise; - /** Called when a bracketed paste contains exactly one image-file path. */ - onPasteImagePath?: (path: string) => void; + /** Called when a bracketed paste contains one or more image-file paths. */ + onPasteImagePath?: (path: string) => void | Promise; /** Called when the configured raw text-paste shortcut is pressed. */ onPasteTextRaw?: () => void; /** Called when the configured dequeue shortcut is pressed. */ @@ -188,9 +244,13 @@ export class CustomEditor extends Editor { return; } - const pastedImagePath = extractBracketedImagePastePath(data); - if (pastedImagePath && this.onPasteImagePath) { - this.onPasteImagePath(pastedImagePath); + const pastedImagePaths = extractBracketedImagePastePaths(data); + if (pastedImagePaths && this.onPasteImagePath) { + void (async () => { + for (const path of pastedImagePaths) { + await this.onPasteImagePath?.(path); + } + })(); return; } diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index 640eefb30..5ae666fb8 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -191,7 +191,7 @@ export class InputController { this.ctx.keybindings.getKeys("app.clipboard.pasteImage"), ); this.ctx.editor.onPasteImage = () => this.handleImagePaste(); - this.ctx.editor.onPasteImagePath = path => void this.handleImagePathPaste(path); + this.ctx.editor.onPasteImagePath = path => this.handleImagePathPaste(path); this.ctx.editor.setActionKeys( "app.clipboard.pasteTextRaw", this.ctx.keybindings.getKeys("app.clipboard.pasteTextRaw"), diff --git a/packages/coding-agent/test/custom-editor-keybindings.test.ts b/packages/coding-agent/test/custom-editor-keybindings.test.ts index 9009f86c2..767a6c54e 100644 --- a/packages/coding-agent/test/custom-editor-keybindings.test.ts +++ b/packages/coding-agent/test/custom-editor-keybindings.test.ts @@ -1,5 +1,9 @@ import { describe, expect, it, vi } from "bun:test"; -import { CustomEditor, extractBracketedImagePastePath } from "@oh-my-pi/pi-coding-agent/modes/components/custom-editor"; +import { + CustomEditor, + extractBracketedImagePastePath, + extractBracketedImagePastePaths, +} from "@oh-my-pi/pi-coding-agent/modes/components/custom-editor"; import { defaultEditorTheme } from "../../tui/test/test-themes"; function ctrl(key: string): string { @@ -24,7 +28,9 @@ describe("CustomEditor bracketed image path paste", () => { it("routes a single pasted image path to the image-path handler", () => { const editor = createEditor(); const paths: string[] = []; - editor.onPasteImagePath = path => paths.push(path); + editor.onPasteImagePath = path => { + paths.push(path); + }; editor.handleInput("\x1b[200~/tmp/screenshot.png\x1b[201~"); @@ -32,6 +38,33 @@ describe("CustomEditor bracketed image path paste", () => { expect(editor.getText()).toBe(""); }); + it("routes multiple pasted image paths to the image-path handler in order", async () => { + const editor = createEditor(); + const paths: string[] = []; + editor.onPasteImagePath = path => { + paths.push(path); + }; + + editor.handleInput("\x1b[200~/tmp/first.png /tmp/second.webp\x1b[201~"); + await Promise.resolve(); + + expect(paths).toEqual(["/tmp/first.png", "/tmp/second.webp"]); + expect(editor.getText()).toBe(""); + }); + + it("keeps spaces inside pasted image paths when splitting a multi-image paste", () => { + expect( + extractBracketedImagePastePaths("\x1b[200~/tmp/My First Screenshot.png /tmp/second image.jpg\x1b[201~"), + ).toEqual(["/tmp/My First Screenshot.png", "/tmp/second image.jpg"]); + }); + + it("unescapes shell-escaped spaces in pasted image paths", () => { + expect(extractBracketedImagePastePaths("\x1b[200~/tmp/My\\ First.png /tmp/second.gif\x1b[201~")).toEqual([ + "/tmp/My First.png", + "/tmp/second.gif", + ]); + }); + it("leaves ordinary bracketed paste text on the editor path", () => { expect(extractBracketedImagePastePath("\x1b[200~not an image.txt\x1b[201~")).toBeUndefined(); }); diff --git a/packages/tui/src/components/image.ts b/packages/tui/src/components/image.ts index 5a0fb31b5..564a46857 100644 --- a/packages/tui/src/components/image.ts +++ b/packages/tui/src/components/image.ts @@ -27,6 +27,8 @@ export interface ImageOptions { const EMPTY_IDS: readonly number[] = []; const EMPTY_TRANSMITS: readonly string[] = []; +const RESERVED_IMAGE_ROW = "\x1b[0m"; + /** Default count of inline images kept as live graphics before older ones fall back to text. */ export const DEFAULT_MAX_INLINE_IMAGES = 8; @@ -188,6 +190,10 @@ export class ImageBudget { this.#pendingTransmits.push(sequence); } + hasPendingTransmits(): boolean { + return this.#pendingTransmits.length > 0; + } + /** Transmit sequences to write before this frame's placements; clears the queue. */ takeTransmits(): readonly string[] { if (this.#pendingTransmits.length === 0) return EMPTY_TRANSMITS; @@ -296,7 +302,7 @@ export class Image implements Component { // moves the cursor back up, then emits the image sequence. lines = []; for (let i = 0; i < result.rows - 1; i++) { - lines.push(""); + lines.push(RESERVED_IMAGE_ROW); } const moveUp = result.rows > 1 ? `\x1b[${result.rows - 1}A` : ""; lines.push(moveUp + (result.sequence ?? "")); diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 93675f9ce..a6cc4552e 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -13,7 +13,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { performance } from "node:perf_hooks"; -import { $flag, getDebugLogPath } from "@oh-my-pi/pi-utils"; +import { $flag, getDebugLogPath, isBunTestRuntime } from "@oh-my-pi/pi-utils"; import { DEFAULT_MAX_INLINE_IMAGES, ImageBudget } from "./components/image"; import { planDeccaraFills } from "./deccara"; import { isKeyRelease, matchesKey } from "./keys"; @@ -494,6 +494,10 @@ export class TUI extends Container { // arrives (issue #2088). Coalescing every SIGWINCH inside this window into // a single forced render lets the multiplexer settle first. static readonly #MULTIPLEXER_RESIZE_DEBOUNCE_MS = 50; + // Ghostty can drop Kitty graphics commands sent during its first post-startup + // settle window, leaving only Unicode placeholder cells. Hold the first image + // paint until that window has passed; later images render normally. + static readonly #GHOSTTY_INITIAL_IMAGE_DELAY_MS = 100; // Post-paint settle window for ConPTY hosts. The `sessionReplace` / // `historyRebuild` / `overlayRebuild` intents drive `#emitFullPaint` over // a transcript that overflows the viewport, scroll-pushing everything past @@ -560,6 +564,9 @@ export class TUI extends Container { // Caps how many inline images render as live graphics; older ones fall back // to text via a purge + full redraw. Cap is configured by the host app. #imageBudget = new ImageBudget(DEFAULT_MAX_INLINE_IMAGES, () => this.requestRender()); + #ghosttyInitialImageDelayDone = false; + #ghosttyInitialImageDelayTimer: RenderTimer | undefined; + #ghosttyImageReadyAtMs = 0; #clearScrollbackOnNextRender = false; #forceViewportRepaintOnNextRender = false; #allowUnknownViewportMutationOnNextRender = false; @@ -889,6 +896,8 @@ export class TUI extends Container { start(options?: TUIStartOptions): void { this.#stopped = false; + this.#ghosttyInitialImageDelayDone = false; + this.#ghosttyImageReadyAtMs = this.#renderScheduler.now() + TUI.#GHOSTTY_INITIAL_IMAGE_DELAY_MS; // A DECRQM report for mode 2026 is authoritative: enable synchronized // output when the terminal reports support (upgrading conservatively // defaulted-off hosts like zellij/tmux-master/foot) and disable it when @@ -1127,6 +1136,10 @@ export class TUI extends Container { this.#renderTimer.cancel(); this.#renderTimer = undefined; } + if (this.#ghosttyInitialImageDelayTimer) { + this.#ghosttyInitialImageDelayTimer.cancel(); + this.#ghosttyInitialImageDelayTimer = undefined; + } if (this.#multiplexerResizeTimer) { this.#multiplexerResizeTimer.cancel(); this.#multiplexerResizeTimer = undefined; @@ -1371,6 +1384,36 @@ export class TUI extends Container { } this.#postFullPaintSettleUntilMs = 0; } + + #maybeDeferGhosttyInitialImagePaint(): boolean { + if (this.#ghosttyInitialImageDelayDone) return false; + if (isBunTestRuntime()) { + this.#ghosttyInitialImageDelayDone = true; + return false; + } + if (TERMINAL.id !== "ghostty" || TERMINAL.imageProtocol !== ImageProtocol.Kitty) { + this.#ghosttyInitialImageDelayDone = true; + return false; + } + if (!this.#imageBudget.hasPendingTransmits()) return false; + if (this.#ghosttyInitialImageDelayTimer) return true; + + const delayMs = Math.max(0, this.#ghosttyImageReadyAtMs - this.#renderScheduler.now()); + if (delayMs === 0) { + this.#ghosttyInitialImageDelayDone = true; + return false; + } + + this.#ghosttyInitialImageDelayTimer = this.#renderScheduler.scheduleRender(() => { + this.#ghosttyInitialImageDelayTimer = undefined; + this.#ghosttyInitialImageDelayDone = true; + if (this.#stopped) return; + this.#lastRenderAt = this.#renderScheduler.now(); + this.#doRender(); + if (this.#renderRequested) this.#scheduleRender(); + }, delayMs); + return true; + } #prepareForcedRender(clearScrollback: boolean): void { const geometryChanged = (this.#previousWidth > 0 && this.#previousWidth !== this.terminal.columns) || @@ -1984,6 +2027,7 @@ export class TUI extends Container { // paints, so subsequent frames re-emit only the tiny placement sequence. // `a=t` produces no display, so writing it ahead of the synchronized paint // is artifact-free. + if (this.#maybeDeferGhosttyInitialImagePaint()) return; const imageTransmits = this.#imageBudget.takeTransmits(); if (imageTransmits.length > 0) { let transmitBuffer = ""; From 42a05f7b09133322e73600bb3fa1e512ea6a6097 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:47:02 +0200 Subject: [PATCH 22/29] fix(coding-agent/modes): allowed transcript rows to re-earn append-only after clean frames - Changed transcript commit tracking from a volatile boolean to a cooldown counter that decays over clean frames. - Normalized row equality checks to ignore ANSI and trailing-space noise so equivalent renders do not trigger rewrites. - Removed the Bun test-runtime exception from Ghostty image paint deferral so normal delay logic always runs. --- .../modes/components/transcript-container.ts | 94 ++++++++++++++----- packages/tui/src/components/image.ts | 4 +- packages/tui/src/tui.ts | 6 +- packages/tui/test/image-budget.test.ts | 55 ++++++++++- packages/tui/test/image-render.test.ts | 1 + 5 files changed, 129 insertions(+), 31 deletions(-) diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index 66c1910aa..89c2c5185 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -7,7 +7,11 @@ interface FrozenRender { lines: string[]; generation: number; appendOnly: boolean; - volatile: boolean; + /** + * Frames remaining until a block that rewrote an interior row may re-earn + * append-only status. `0` means the block is not under rewrite suspicion. + */ + volatileCooldown: number; } interface SnapshotCarrier { @@ -51,10 +55,41 @@ function stripPlainBlankEdges(lines: string[]): string[] { interface LiveCommitState { appendOnly: boolean; - volatile: boolean; + volatileCooldown: number; safeLength: number; } +/** + * Render frames a block must stay clean (static or append-shaped) after an + * interior rewrite before its rows become committable again. A one-off + * re-layout (a codespan finalizing across a wrap boundary, a paragraph + * re-parsed as a heading) only suspends commits briefly — the pinned emitter + * appends from the stalled high-water mark, so the gap backfills contiguously + * once the block re-earns append-only. Periodic animations (a spinner rewrites + * its row every few frames) keep resetting the countdown and never re-earn it, + * so genuinely volatile blocks stay deferred. Frames arrive at most at the + * TUI's 30 Hz render cadence, so 30 frames ≈ 1s of clean streaming. + */ +const VOLATILE_REARM_FRAMES = 30; + +/** + * Visible-content form of a row: SGR/OSC bytes and trailing pad spaces are + * write framing, not content. A styled line's closing escape moves when the + * line stops being the last of its span (a wrapped thinking paragraph growing + * by one row), and width-padded rows shift their trailing spaces as text + * grows; both leave the on-screen cells identical and must not count as a + * rewrite of a committed-candidate row. Committed scrollback rows are written + * with a full SGR/OSC reset terminator, so escape-placement drift between + * visually identical renders cannot bleed styles across rows. + */ +function normalizeRow(line: string): string { + return Bun.stripANSI(line).trimEnd(); +} + +function rowsVisiblyEqual(prev: string, cur: string): boolean { + return prev === cur || normalizeRow(prev) === normalizeRow(cur); +} + function hasValidSnapshot( snapshot: FrozenRender | undefined, width: number, @@ -66,14 +101,14 @@ function hasValidSnapshot( function commonPrefixLength(prev: string[], cur: string[]): number { const limit = Math.min(prev.length, cur.length); let i = 0; - while (i < limit && prev[i] === cur[i]) i++; + while (i < limit && rowsVisiblyEqual(prev[i]!, cur[i]!)) i++; return i; } function commonSuffixLength(prev: string[], cur: string[], prefixLength: number): number { const limit = Math.min(prev.length - prefixLength, cur.length - prefixLength); let i = 0; - while (i < limit && prev[prev.length - 1 - i] === cur[cur.length - 1 - i]) i++; + while (i < limit && rowsVisiblyEqual(prev[prev.length - 1 - i]!, cur[cur.length - 1 - i]!)) i++; return i; } @@ -84,42 +119,53 @@ function deriveLiveCommitState( generation: number, ): LiveCommitState { let appendOnly = false; - let volatile = false; + let volatileCooldown = 0; if (hasValidSnapshot(previous, width, generation)) { appendOnly = previous.appendOnly; - volatile = previous.volatile; + volatileCooldown = previous.volatileCooldown; const prefixLength = commonPrefixLength(previous.lines, current); const staticRender = prefixLength === previous.lines.length && prefixLength === current.length; + let cleanFrame = true; if (!staticRender) { const suffixLength = commonSuffixLength(previous.lines, current, prefixLength); // Append-only growth never rewrites a row that may already have scrolled - // into native scrollback; it only grows the block at/near its tail. Three + // into native scrollback; it only grows the block at/near its tail. Four // shapes qualify: a pure bottom append, an insertion above stable trailing - // chrome (a streaming tool's footer/border), and an in-place extension of - // the current line by one streamed token (line count unchanged). The first - // two preserve every previous row across a matching prefix + suffix; the - // last leaves a single divergent previous row that the current row merely - // lengthens. A divergent interior row that is genuinely rewritten means the - // block re-laid-out committed content — volatile, and never committed. + // chrome (a streaming tool's footer/border), an in-place extension of the + // current line by one streamed token (line count unchanged), and a + // wrap-shrink of the current line where its last word grew past the wrap + // column and moved down onto an appended row. The first two preserve every + // previous row across a matching prefix + suffix; the last two leave a + // single divergent previous row — the block's in-flight bottom line, which + // cannot have been committed (commits stop at the viewport top and the + // bottom line is by definition on screen). Any other divergent interior + // row means the block re-laid-out committed-candidate content — a rewrite, + // which suspends commits until the block re-earns append-only. const preservedEveryRow = prefixLength + suffixLength >= previous.lines.length; - const tailExtendedInPlace = - prefixLength + suffixLength === previous.lines.length - 1 && - prefixLength < current.length && - current[prefixLength]!.startsWith(previous.lines[prefixLength]!); - if ((preservedEveryRow || tailExtendedInPlace) && current.length >= previous.lines.length && !volatile) { - appendOnly = true; - } else if (!preservedEveryRow && !tailExtendedInPlace) { - volatile = true; + let tailExtendedInPlace = false; + if (!preservedEveryRow && prefixLength + suffixLength === previous.lines.length - 1 && prefixLength < current.length) { + const prevTail = normalizeRow(previous.lines[prefixLength]!); + const curTail = normalizeRow(current[prefixLength]!); + tailExtendedInPlace = + curTail.startsWith(prevTail) || + (current.length > previous.lines.length && prevTail.startsWith(curTail)); + } + if ((preservedEveryRow || tailExtendedInPlace) && current.length >= previous.lines.length) { + if (volatileCooldown === 0) appendOnly = true; + } else { + cleanFrame = false; appendOnly = false; + volatileCooldown = VOLATILE_REARM_FRAMES; } } + if (cleanFrame && volatileCooldown > 0) volatileCooldown--; } return { appendOnly, - volatile, - safeLength: volatile ? 0 : appendOnly ? current.length : 0, + volatileCooldown, + safeLength: appendOnly ? current.length : 0, }; } @@ -265,7 +311,7 @@ export class TranscriptContainer extends Container implements NativeScrollbackLi lines: contribution, generation: this.#generation, appendOnly: liveCommitState?.appendOnly ?? false, - volatile: liveCommitState?.volatile ?? false, + volatileCooldown: liveCommitState?.volatileCooldown ?? 0, }; } } diff --git a/packages/tui/src/components/image.ts b/packages/tui/src/components/image.ts index 564a46857..e4695564d 100644 --- a/packages/tui/src/components/image.ts +++ b/packages/tui/src/components/image.ts @@ -27,9 +27,10 @@ export interface ImageOptions { const EMPTY_IDS: readonly number[] = []; const EMPTY_TRANSMITS: readonly string[] = []; +// Direct placements reserve height with leading zero-width rows. Keep them +// non-plain so transcript blank-edge trimming does not collapse image-only blocks. const RESERVED_IMAGE_ROW = "\x1b[0m"; - /** Default count of inline images kept as live graphics before older ones fall back to text. */ export const DEFAULT_MAX_INLINE_IMAGES = 8; @@ -190,6 +191,7 @@ export class ImageBudget { this.#pendingTransmits.push(sequence); } + /** Whether a frame has image data queued but not yet written to the terminal. */ hasPendingTransmits(): boolean { return this.#pendingTransmits.length > 0; } diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index a6cc4552e..6f7ab6da4 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -13,7 +13,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { performance } from "node:perf_hooks"; -import { $flag, getDebugLogPath, isBunTestRuntime } from "@oh-my-pi/pi-utils"; +import { $flag, getDebugLogPath } from "@oh-my-pi/pi-utils"; import { DEFAULT_MAX_INLINE_IMAGES, ImageBudget } from "./components/image"; import { planDeccaraFills } from "./deccara"; import { isKeyRelease, matchesKey } from "./keys"; @@ -1387,10 +1387,6 @@ export class TUI extends Container { #maybeDeferGhosttyInitialImagePaint(): boolean { if (this.#ghosttyInitialImageDelayDone) return false; - if (isBunTestRuntime()) { - this.#ghosttyInitialImageDelayDone = true; - return false; - } if (TERMINAL.id !== "ghostty" || TERMINAL.imageProtocol !== ImageProtocol.Kitty) { this.#ghosttyInitialImageDelayDone = true; return false; diff --git a/packages/tui/test/image-budget.test.ts b/packages/tui/test/image-budget.test.ts index 57504afd6..f6b9d4c54 100644 --- a/packages/tui/test/image-budget.test.ts +++ b/packages/tui/test/image-budget.test.ts @@ -19,7 +19,7 @@ import { } from "@oh-my-pi/pi-tui/terminal-capabilities"; import { VirtualTerminal } from "./virtual-terminal"; -type MutableTerminalInfo = { imageProtocol: ImageProtocol | null }; +type MutableTerminalInfo = { id: string; imageProtocol: ImageProtocol | null }; const terminal = TERMINAL as unknown as MutableTerminalInfo; const BASE64_ONE_PIXEL_PNG = @@ -436,6 +436,59 @@ describe("TUI inline-image budget", () => { tui.stop(); } }); + + it("holds the first Ghostty image paint until the startup settle window passes", () => { + const originalId = terminal.id; + const originalGraphics = { ...getKittyGraphics() }; + const term = new VirtualTerminal(40, 12); + const writes: string[] = []; + const realWrite = term.write.bind(term); + vi.spyOn(term, "write").mockImplementation((data: string) => { + writes.push(data); + realWrite(data); + }); + + let now = 0; + const scheduled: Array<{ delayMs: number; callback: () => void; canceled: boolean }> = []; + const renderScheduler = { + now: () => now, + scheduleImmediate: (callback: () => void) => callback(), + scheduleRender: (callback: () => void, delayMs: number) => { + const entry = { delayMs, callback, canceled: false }; + scheduled.push(entry); + return { + cancel: () => { + entry.canceled = true; + }, + }; + }, + }; + + terminal.id = "ghostty"; + terminal.imageProtocol = ImageProtocol.Kitty; + setKittyGraphics({ unicodePlaceholders: true }); + + const tui = new TUI(term, undefined, { renderScheduler }); + tui.addChild(makeImage(tui.imageBudget, "only")); + + try { + tui.start(); + expect(writes.join("")).not.toContain("\x1b_Ga=t"); + + const delayed = scheduled.find(entry => !entry.canceled && entry.delayMs === 100); + expect(delayed).toBeDefined(); + now = 100; + delayed?.callback(); + + const output = writes.join(""); + expect(output).toContain("\x1b_Ga=t"); + expect(output).toContain(BASE64_ONE_PIXEL_PNG); + } finally { + tui.stop(); + terminal.id = originalId; + setKittyGraphics(originalGraphics); + } + }); }); describe("kitty transmit / placement encoding", () => { diff --git a/packages/tui/test/image-render.test.ts b/packages/tui/test/image-render.test.ts index 4358373a8..0a8a9c547 100644 --- a/packages/tui/test/image-render.test.ts +++ b/packages/tui/test/image-render.test.ts @@ -118,6 +118,7 @@ describe("terminal image rendering", () => { const lines = image.render(20); + expect(lines[0]).toBe("\x1b[0m"); expect(lines).toHaveLength(2); expect(lines[1]).toContain("\x1b[1A"); expect(lines[1]).toContain("c=2"); From 28a19e91dbd0829ce7fe899611e3e508622a7221 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:48:49 +0200 Subject: [PATCH 23/29] fix(packages/coding-agent): addressed transcript append-only scrolling - Added regressions for transcript append-only handling on wrapped styled rows and trailing-line shrink. - Added streaming-thinking and spinner-style reproduction coverage for commit-safe boundaries. --- .../modes/components/transcript-container.ts | 9 +- .../scratch-thinking-commit-repro.test.ts | 80 ++++++++++ .../test/tool-live-region-scrollback.test.ts | 141 ++++++++++++++++++ 3 files changed, 227 insertions(+), 3 deletions(-) create mode 100644 packages/coding-agent/test/scratch-thinking-commit-repro.test.ts diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index 89c2c5185..5e6a36262 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -144,12 +144,15 @@ function deriveLiveCommitState( // which suspends commits until the block re-earns append-only. const preservedEveryRow = prefixLength + suffixLength >= previous.lines.length; let tailExtendedInPlace = false; - if (!preservedEveryRow && prefixLength + suffixLength === previous.lines.length - 1 && prefixLength < current.length) { + if ( + !preservedEveryRow && + prefixLength + suffixLength === previous.lines.length - 1 && + prefixLength < current.length + ) { const prevTail = normalizeRow(previous.lines[prefixLength]!); const curTail = normalizeRow(current[prefixLength]!); tailExtendedInPlace = - curTail.startsWith(prevTail) || - (current.length > previous.lines.length && prevTail.startsWith(curTail)); + curTail.startsWith(prevTail) || (current.length > previous.lines.length && prevTail.startsWith(curTail)); } if ((preservedEveryRow || tailExtendedInPlace) && current.length >= previous.lines.length) { if (volatileCooldown === 0) appendOnly = true; diff --git a/packages/coding-agent/test/scratch-thinking-commit-repro.test.ts b/packages/coding-agent/test/scratch-thinking-commit-repro.test.ts new file mode 100644 index 000000000..5689c41b9 --- /dev/null +++ b/packages/coding-agent/test/scratch-thinking-commit-repro.test.ts @@ -0,0 +1,80 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message"; +import { TranscriptContainer } from "@oh-my-pi/pi-coding-agent/modes/components/transcript-container"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { TERMINAL } from "@oh-my-pi/pi-tui"; + +type MutableTerminalInfo = { eagerEraseScrollbackRisk: boolean }; +const mutableTerminalInfo = TERMINAL as unknown as MutableTerminalInfo; + +function makeThinkingMessage(thinking: string): AssistantMessage { + return { + role: "assistant", + content: [{ type: "thinking", thinking }], + api: "anthropic", + provider: "anthropic", + model: "test-model", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: Date.now(), + }; +} + +const PARA = (n: number) => + `I'm realizing paragraph ${n} that thinking inference only happens when model.reasoning is enabled, so for a live-discovered model to get adaptive thinking configured, it needs reasoning true set upfront. That value comes from the defaults in the OpenAI-compatible models fetch, which gets applied when mapping a model without an existing reference.`; + +describe("scratch: streaming thinking commit boundary", () => { + beforeAll(async () => { + await initTheme(); + await Settings.init({ inMemory: true, cwd: process.cwd() }); + }); + + it("traces commitSafeEnd while streaming styled thinking", () => { + const saved = TERMINAL.eagerEraseScrollbackRisk; + mutableTerminalInfo.eagerEraseScrollbackRisk = true; + try { + const chat = new TranscriptContainer(); + const component = new AssistantMessageComponent(undefined, false); + chat.addChild(component); + + const fullText = [PARA(1), PARA(2), PARA(3)].join("\n\n"); + const words = fullText.split(" "); + + let prevRows: string[] = []; + let firstStall: number | undefined; + // stream ~5 words per frame (30fps coalescing) + for (let i = 5; i <= words.length; i += 5) { + const text = words.slice(0, i).join(" "); + component.updateContent(makeThinkingMessage(text)); + const rows = chat.render(120); + const safeEnd = chat.getNativeScrollbackCommitSafeEnd(); + if (safeEnd === undefined && rows.length > 3 && firstStall === undefined) { + firstStall = i; + // dump the frame diff that broke append-only detection + console.log(`--- stall at word ${i}, rows=${rows.length}, prevRows=${prevRows.length}`); + const limit = Math.max(prevRows.length, rows.length); + for (let r = 0; r < limit; r++) { + if (prevRows[r] !== rows[r]) { + console.log(` row ${r} prev: ${JSON.stringify(prevRows[r])}`); + console.log(` row ${r} cur : ${JSON.stringify(rows[r])}`); + } + } + } + prevRows = rows; + } + console.log(`total words=${words.length}, firstStall=${firstStall}`); + expect(firstStall).toBeUndefined(); + } finally { + mutableTerminalInfo.eagerEraseScrollbackRisk = saved; + } + }); +}); diff --git a/packages/coding-agent/test/tool-live-region-scrollback.test.ts b/packages/coding-agent/test/tool-live-region-scrollback.test.ts index 061701837..8caf5deee 100644 --- a/packages/coding-agent/test/tool-live-region-scrollback.test.ts +++ b/packages/coding-agent/test/tool-live-region-scrollback.test.ts @@ -103,6 +103,85 @@ describe("transcript reactive commit boundary", () => { expect(chat.getNativeScrollbackCommitSafeEnd()).toBeUndefined(); }); }); + + it("treats escape placement and pad drift on visually unchanged rows as append-only", async () => { + await withTerminalRisk(true, () => { + const chat = new TranscriptContainer(); + // Field failure shape (streaming styled thinking): the previous last row + // carried the span-closing SGR before its width padding; when the + // paragraph wrapped onto a new row, the close moved to the new last row + // while the first row's visible cells stayed identical. + const sty = "\x1b[38;2;156;163;176m"; + const block = new MutableLiveBlock([`${sty}alpha beta\x1b[39m `]); + chat.addChild(block); + + chat.render(80); + block.setLines([`${sty}alpha beta `, `${sty}gamma\x1b[39m `]); + chat.render(80); + expect(chat.getNativeScrollbackCommitSafeEnd()).toBe(2); + }); + }); + + it("treats a wrap-shrink of the trailing line as append-only", async () => { + await withTerminalRisk(true, () => { + const chat = new TranscriptContainer(); + // A streamed token extends the last word past the wrap column, so the + // word moves down onto an appended row and the previous bottom line + // shrinks. The bottom line is on screen by definition, so this is not a + // rewrite of committed-candidate rows. + const block = new MutableLiveBlock(["para one", "foo bar baz"]); + chat.addChild(block); + + chat.render(80); + block.setLines(["para one", "foo bar", "bazqux and more"]); + chat.render(80); + expect(chat.getNativeScrollbackCommitSafeEnd()).toBe(3); + }); + }); + + it("re-earns append-only after a one-off interior rewrite heals", async () => { + await withTerminalRisk(true, () => { + const chat = new TranscriptContainer(); + const block = new MutableLiveBlock(["top", "old", "bottom"]); + chat.addChild(block); + + chat.render(80); + // Interior rewrite (a codespan finalizing across a wrap) suspends commits. + block.setLines(["top", "new", "bottom"]); + chat.render(80); + expect(chat.getNativeScrollbackCommitSafeEnd()).toBeUndefined(); + + // Clean static frames re-arm the block... + for (let i = 0; i < 30; i++) chat.render(80); + // ...and the next append-shaped frame resumes committing the full block, + // so the pinned emitter can backfill the stalled gap contiguously. + block.setLines(["top", "new", "bottom", "appended"]); + chat.render(80); + expect(chat.getNativeScrollbackCommitSafeEnd()).toBe(4); + }); + }); + + it("keeps a periodically rewriting block (spinner) deferred", async () => { + await withTerminalRisk(true, () => { + const chat = new TranscriptContainer(); + const block = new MutableLiveBlock(["⠋ running", "body"]); + chat.addChild(block); + + chat.render(80); + const glyphs = ["⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏", "⠋"]; + for (const glyph of glyphs) { + // Spinner advances every third frame; the static frames in between + // must never accumulate into a re-arm. + block.setLines([`${glyph} running`, "body"]); + chat.render(80); + chat.render(80); + chat.render(80); + } + block.setLines(["⠋ running", "body", "appended"]); + chat.render(80); + expect(chat.getNativeScrollbackCommitSafeEnd()).toBeUndefined(); + }); + }); }); describe("tool live-region scrollback", () => { @@ -480,6 +559,12 @@ function makeAssistantMessage(text: string): AssistantMessage { }; } +function makeThinkingMessage(thinking: string): AssistantMessage { + const message = makeAssistantMessage(""); + message.content = [{ type: "thinking", thinking }]; + return message; +} + describe("assistant live-region scrollback", () => { beforeAll(async () => { await initTheme(); @@ -533,4 +618,60 @@ describe("assistant live-region scrollback", () => { } }); }); + + it("commits scrolled-off styled thinking paragraphs to scrollback while streaming", async () => { + if (process.platform === "win32") return; + + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(120, 12); + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => + undefined; + const tui = new TUI(term); + const chat = new TranscriptContainer(); + const component = new AssistantMessageComponent(undefined, false); + // Word-wrapped italic/colored paragraphs — the styled streaming shape the + // raw-byte append detector mis-classified as volatile (the span-closing + // SGR moves rows as the paragraph wraps), which froze the commit boundary + // and dropped every later paragraph that scrolled past the viewport top. + const paragraphs = Array.from( + { length: 8 }, + (_unused, i) => + `PARA-${i} considering the resolver path and the descriptor defaults, the policy layer must keep the ` + + `reasoning flag intact while discovery maps an unknown model entry onto the bundled reference shape ` + + `so the runtime request stays correct across upstream metadata shifts.`, + ); + const fullText = paragraphs.join("\n\n"); + const words = fullText.split(" "); + + try { + chat.addChild(component); + tui.addChild(chat); + tui.start(); + tui.setEagerNativeScrollbackRebuild(true); + await term.waitForRender(); + + // Stream a few words per frame so the in-flight bottom line extends, + // wraps, and sheds words onto new rows across many coalesced frames. + for (let i = 5; i <= words.length; i += 5) { + component.updateContent(makeThinkingMessage(words.slice(0, i).join(" "))); + tui.requestRender(); + await term.waitForRender(); + } + + const scrollText = stripRows(term.getScrollBuffer()); + const viewportText = stripRows(term.getViewport()); + + // Early paragraphs scrolled above the viewport: they must live in + // native scrollback, not vanish into the dropped gap. + expect(viewportText).not.toContain("PARA-0"); + expect(scrollText).toContain("PARA-0"); + expect(scrollText).toContain("PARA-4"); + // The tail is still on screen. + expect(viewportText).toContain("PARA-7"); + } finally { + tui.stop(); + await term.flush(); + } + }); + }); }); From 7d73c88133d89c1ad5ba310418b7f9fcf7d61286 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:48:49 +0200 Subject: [PATCH 24/29] feat(packages/ai): added fable mythos support with adaptive tool policy - Added Anthropic Fable and Mythos support by updating model kinds, parsing, and checks. - Set Fable/Mythos compatibility and policy values, including 1M context, 128k max tokens, and costs. - Enabled adaptive thinking display for claude-fable-5 and claude-mythos-5 and kept output effort low when disabled. - Adjusted forced tool-choice and cache-control behavior for unsupported features and empty text content. --- packages/ai/src/model-thinking.ts | 52 +++++++++++++---- packages/ai/src/providers/amazon-bedrock.ts | 31 +++++----- packages/ai/src/providers/anthropic.ts | 57 +++++++++++++------ .../ai/src/providers/openai-completions.ts | 7 ++- packages/ai/src/types.ts | 15 ++++- 5 files changed, 115 insertions(+), 47 deletions(-) diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 3758c6d6c..8913a6142 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -23,7 +23,7 @@ type SemVer = { }; type GeminiKind = "pro" | "flash"; -type AnthropicKind = "opus" | "sonnet"; +type AnthropicKind = "opus" | "sonnet" | "fable" | "mythos"; type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "mini" | "max" | "nano"; const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial> = { @@ -308,7 +308,8 @@ export function mapEffortToAnthropicAdaptiveEffort( ): "low" | "medium" | "high" | "xhigh" | "max" { const supported = requireSupportedEffort(model, effort); if (anthropicModelHasRealXHighEffort(model)) { - // Opus 4.7+ on the Messages API exposes the full five-tier adaptive scale + // Opus 4.7+ and Fable/Mythos 5 on the Messages API expose the full + // five-tier adaptive scale // (low/medium/high/xhigh/max). Shift our user-facing efforts up one notch so // the top tier reaches the genuine "max" and "high" lands on Anthropic's // recommended "xhigh" coding/agentic default. @@ -341,33 +342,44 @@ export function mapEffortToAnthropicAdaptiveEffort( } /** - * Returns true for Anthropic models with Opus 4.7 API restrictions: + * Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions: * - Sampling parameters (temperature/top_p/top_k) return 400 error * - Thinking content is omitted by default (needs display: "summarized") */ export function hasOpus47ApiRestrictions(modelId: string): boolean { const parsed = parseAnthropicModel(getCanonicalModelId(modelId)); if (!parsed) return false; - return semverGte(parsed.version, "4.7") && parsed.kind === "opus"; + return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind); } /** * Mid-conversation `role: "system"` messages (system instructions appended at * non-first positions in the `messages` array) are supported starting with - * Claude Opus 4.8. Earlier Claude models reject the role. + * Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude + * models reject the role. * @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages */ export function supportsMidConversationSystemMessages(modelId: string): boolean { const parsed = parseAnthropicModel(getCanonicalModelId(modelId)); if (!parsed) return false; - return parsed.kind === "opus" && semverGte(parsed.version, "4.8"); + return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind); +} + +export function isAnthropicFableOrMythosModel(modelId: string): boolean { + const parsed = parseAnthropicModel(getCanonicalModelId(modelId)); + return parsed !== null && isFableOrMythos(parsed.kind); +} + +function isFableOrMythos(kind: AnthropicKind): boolean { + return kind === "fable" || kind === "mythos"; } function anthropicModelHasRealXHighEffort(model: ApiModel): boolean { if (model.api !== "anthropic-messages") return false; const parsedModel = parseKnownModel(model.id); - if (parsedModel.family !== "anthropic" || parsedModel.kind !== "opus") return false; - return semverGte(parsedModel.version, "4.7"); + if (parsedModel.family !== "anthropic") return false; + if (isFableOrMythos(parsedModel.kind)) return true; + return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7"); } function applyGeneratedModelPolicy(model: ApiModel): void { @@ -431,6 +443,19 @@ function applyAnthropicCatalogPolicy(model: ApiModel, parsedModel: Anthropi model.contextWindow = 1000000; model.maxTokens = 128000; } + + // Claude Fable/Mythos 5: Anthropic's /v1/models omits token limits and + // pricing, and models.dev lags new releases. Pin authoritative values from + // the model card (1M context / 128k output) and pricing docs ($10 in / $50 + // out per MTok). + if (model.provider === "anthropic" && isFableOrMythos(parsedModel.kind)) { + model.contextWindow = 1_000_000; + model.maxTokens = 128_000; + model.cost.input = 10; + model.cost.output = 50; + model.cost.cacheRead = 1; + model.cost.cacheWrite = 12.5; + } } function inferGeneratedApplyPatchToolType( @@ -562,7 +587,9 @@ function inferAnthropicSupportedEfforts( (model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") && semverGte(parsedModel.version, "4.6") ) { - return parsedModel.kind === "opus" ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS; + return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind) + ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH + : DEFAULT_REASONING_EFFORTS; } return inferFallbackEfforts(model); } @@ -618,7 +645,10 @@ function inferThinkingControlMode( case "bedrock-converse-stream": if (parsedModel.family === "anthropic") { - if (semverGte(parsedModel.version, "4.6") && parsedModel.kind === "opus") { + if ( + semverGte(parsedModel.version, "4.6") && + (parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)) + ) { return "anthropic-adaptive"; } if (semverGte(parsedModel.version, "4.5")) { @@ -658,7 +688,7 @@ function parseGeminiModel(modelId: string): GeminiModel | null { } function parseAnthropicModel(modelId: string): AnthropicModel | null { - const match = /claude-(opus|sonnet)-(\d{1,2}(?:[.-]\d{1,2}){0,2})\b/.exec(modelId); + const match = /claude-(opus|sonnet|fable|mythos)-(\d{1,2}(?:[.-]\d{1,2}){0,2})\b/.exec(modelId); if (!match) { return null; } diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 49b1523aa..1197fa8a9 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -57,11 +57,12 @@ export interface BedrockOptions extends StreamOptions { * - `"omitted"`: thinking content is suppressed; the encrypted signature still * travels back for multi-turn continuity. * - * Starting with Claude Opus 4.7 the Anthropic API default is `"omitted"`, which - * leaves callers waiting on a silent stream during long reasoning runs (issue - * #1373). We default to `"summarized"` so adaptive-thinking models that accept - * the field keep producing visible thinking deltas. Older adaptive-thinking - * models (Opus 4.6, Sonnet 4.6+) reject the field, so we omit it for them. + * Starting with Claude Opus 4.7 and Claude Fable/Mythos 5 the Anthropic API + * default is `"omitted"`, which leaves callers waiting on a silent stream during + * long reasoning runs (issue #1373). We default to `"summarized"` so adaptive- + * thinking models that accept the field keep producing visible thinking deltas. + * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field, so + * we omit it for them. */ thinkingDisplay?: BedrockThinkingDisplay; } @@ -792,10 +793,11 @@ function buildAdditionalModelRequestFields( const mode = model.thinking?.mode; if (mode === "anthropic-adaptive") { const effort = mapEffortToAnthropicAdaptiveEffort(model, reasoning); - // Starting with Claude Opus 4.7, Anthropic switched the adaptive-thinking - // default to "omitted", which silently suppresses streamed reasoning and - // can read as a stalled stream during long reasoning runs (issue #1373). - // Opt back into "summarized" by default on models that accept the field. + // Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, Anthropic switched + // the adaptive-thinking default to "omitted", which silently suppresses + // streamed reasoning and can read as a stalled stream during long reasoning + // runs (issue #1373). Opt back into "summarized" by default on models that + // accept the field. const adaptive: { type: "adaptive"; display?: BedrockThinkingDisplay } = { type: "adaptive" }; if (supportsAdaptiveThinkingDisplay(model.id)) { adaptive.display = options.thinkingDisplay ?? "summarized"; @@ -832,13 +834,14 @@ function buildAdditionalModelRequestFields( } /** - * Adaptive thinking `display` is supported starting with Claude Opus 4.7. - * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field. - * Bedrock model ids are prefixed with region/inference-profile slugs (e.g. - * `eu.anthropic.claude-opus-4-7-...`); the regex matches the `claude-opus-X-Y` - * fragment regardless of prefix. + * Adaptive thinking `display` is supported starting with Claude Opus 4.7 and + * Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet + * 4.6+) reject the field. Bedrock model ids are prefixed with region/inference- + * profile slugs (e.g. `eu.anthropic.claude-opus-4-7-...`); the regex matches + * the Claude model fragment regardless of prefix. */ function supportsAdaptiveThinkingDisplay(modelId: string): boolean { + if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true; const match = /claude-opus-(\d+)-(\d+)/.exec(modelId); if (!match) return false; const major = Number(match[1]); diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 12adf6df0..4c4f3fd55 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -14,6 +14,7 @@ import { } from "@oh-my-pi/pi-utils"; import { hasOpus47ApiRestrictions, + isAnthropicFableOrMythosModel, mapEffortToAnthropicAdaptiveEffort, supportsMidConversationSystemMessages, } from "../model-thinking"; @@ -283,10 +284,12 @@ const ANTHROPIC_STOP_SEQUENCES_MAX = 4; let warnedStopSequencesTrim = false; /** - * Adaptive thinking `display` is supported starting with Claude Opus 4.7. - * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field. + * Adaptive thinking `display` is supported starting with Claude Opus 4.7 and + * Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet + * 4.6+) reject the field. */ function supportsAdaptiveThinkingDisplay(modelId: string): boolean { + if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true; const match = /claude-opus-(\d+)-(\d+)/.exec(modelId); if (!match) return false; const major = Number(match[1]); @@ -896,17 +899,18 @@ export type AnthropicThinkingDisplay = "summarized" | "omitted"; export interface AnthropicOptions extends StreamOptions { /** * Enable extended thinking. - * For Opus 4.6+: uses adaptive thinking (Claude decides when/how much to think). - * For older models: uses budget-based thinking with thinkingBudgetTokens. + * For adaptive-capable models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5): + * uses adaptive thinking (Claude decides when/how much to think). For older + * models: uses budget-based thinking with thinkingBudgetTokens. */ thinkingEnabled?: boolean; /** * Token budget for extended thinking (older models only). - * Ignored for Opus 4.6+ which uses adaptive thinking. + * Ignored for adaptive-capable models. */ thinkingBudgetTokens?: number; /** - * Effort level for adaptive thinking (Opus 4.6+ only). + * Effort level for adaptive thinking. * Controls how much thinking Claude allocates: * - "max": Always thinks with no constraints * - "high": Always thinks, deep reasoning (default) @@ -1279,8 +1283,9 @@ function getAnthropicCompat( model.compat?.supportsMidConversationSystem ?? // First-party Claude API only. Bedrock/Vertex/Foundry and other // Anthropic-compatible proxies reject the role; gate auto-detection on - // the canonical api.anthropic.com host plus an Opus 4.8+ model id. + // the canonical api.anthropic.com host plus a supported model id. (isAnthropicApiBaseUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id)), + supportsForcedToolChoice: model.compat?.supportsForcedToolChoice ?? !isAnthropicFableOrMythosModel(model.id), }; } @@ -2556,12 +2561,12 @@ function buildParams( const compat = getAnthropicCompat(model); if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" }; - // Starting with Claude Opus 4.7, adaptive thinking content is omitted from the - // response by default. Opt into summarized reasoning so thinking deltas keep - // streaming with human-readable content for callers that rely on it. The - // `display` field is gated strictly on model support: Opus 4.6 / Sonnet 4.6+ - // reject it with a 400, so an explicit `thinkingDisplay` MUST NOT force it onto - // a model that can't accept it (a hidden-thinking toggle must never break the request). + // Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking + // content is omitted from the response by default. Opt into summarized + // reasoning so thinking deltas keep streaming with human-readable content for + // callers that rely on it. The `display` field is gated strictly on model + // support: Opus 4.6 / Sonnet 4.6+ reject it with a 400, so an explicit + // `thinkingDisplay` MUST NOT force it onto a model that can't accept it. if (supportsAdaptiveThinkingDisplay(model.id)) { adaptive.display = options.thinkingDisplay ?? "summarized"; } @@ -2576,7 +2581,17 @@ function buildParams( if (mode === "anthropic-budget-effort" && effort) outputConfigEffort = effort; } } else if (options?.thinkingEnabled === false) { - thinking = { type: "disabled" }; + const compat = getAnthropicCompat(model); + if (model.thinking?.mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { + // Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject + // `thinking.type: "disabled"` — adaptive thinking cannot be switched off. + // Omit the thinking field (the API defaults to adaptive) and pin the + // lowest effort so "thinking off" calls stay cheap instead of failing + // the request with a 400 (a hidden-thinking toggle must never break it). + outputConfigEffort = "low"; + } else { + thinking = { type: "disabled" }; + } } } @@ -2607,7 +2622,7 @@ function buildParams( stream: true, }; - // Opus 4.7+ rejects non-default sampling parameters with 400 error. + // Opus 4.7+ and Fable/Mythos 5 reject non-default sampling parameters with 400 error. const thinkingType = params.thinking?.type; const allowSamplingParams = !hasOpus47ApiRestrictions(model.id) && (thinkingType === undefined || thinkingType === "disabled"); @@ -2645,6 +2660,14 @@ function buildParams( } else { params.tool_choice = options.toolChoice; } + // Claude Fable/Mythos 5 reject forced tool use outright ("tool_choice forces + // tool use is not compatible with this model"). Downgrade any/tool → auto so the + // request succeeds; the tool stays available and the caller's prompt steers + // the model toward it. + const choiceType = params.tool_choice?.type; + if ((choiceType === "any" || choiceType === "tool") && !getAnthropicCompat(model).supportsForcedToolChoice) { + params.tool_choice = { type: "auto" }; + } } disableThinkingIfToolChoiceForced(params); @@ -2718,7 +2741,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul /** * A single Anthropic conversation turn, including the mid-conversation - * `system` role (Opus 4.8+). + * `system` role (Opus 4.8+ and Fable/Mythos 5). */ export type AnthropicMessageParam = MessageParam; @@ -2857,7 +2880,7 @@ export function convertAnthropicMessages( } // Upgrade developer-origin params to mid-conversation `system` messages where - // Anthropic's placement rules allow it (Opus 4.8+ on the first-party API). + // Anthropic's placement rules allow it (Opus 4.8+ / Fable/Mythos 5 on first-party API). // Rules: a system message must immediately follow a `user` turn and must be // the last entry or be followed by an `assistant` turn — never first, and // never consecutive. Requiring the next param to be `assistant` (or absent) diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 4072174c0..661c12704 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1542,6 +1542,7 @@ function maybeAddAnthropicCacheControl(compat: ResolvedOpenAICompat, messages: C const content = msg.content; if (typeof content === "string") { + if (content.trim().length === 0) continue; msg.content = [ Object.assign({ type: "text" as const, text: content }, { cache_control: { type: "ephemeral" } }), ]; @@ -1550,10 +1551,12 @@ function maybeAddAnthropicCacheControl(compat: ResolvedOpenAICompat, messages: C if (!Array.isArray(content)) continue; - // Find last text part and add cache_control + // Find last non-empty text part and add cache_control. Empty assistant + // content is valid for tool-call replay, but Anthropic/OpenRouter reject + // empty text blocks once cache_control turns it into structured content. for (let j = content.length - 1; j >= 0; j--) { const part = content[j]; - if (part?.type === "text") { + if (part?.type === "text" && part.text.trim().length > 0) { Object.assign(part, { cache_control: { type: "ephemeral" } }); return; } diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index a37e1077a..7270420f8 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -831,11 +831,20 @@ export interface AnthropicCompat { supportsLongCacheRetention?: boolean; /** * Whether mid-conversation `role: "system"` messages are accepted in the - * `messages` array (Claude Opus 4.8+ on the first-party Claude API and - * Claude Platform on AWS). When unset, auto-detected from the model id and - * base URL. Not available on Bedrock, Vertex AI, or Microsoft Foundry. + * `messages` array (Claude Opus 4.8+ and Claude Fable/Mythos 5 on the + * first-party Claude API and Claude Platform on AWS). When unset, + * auto-detected from the model id and base URL. Not available on Bedrock, + * Vertex AI, or Microsoft Foundry. */ supportsMidConversationSystem?: boolean; + /** + * Whether the model accepts a forced `tool_choice` (`{ type: "any" }` or + * `{ type: "tool", name }`). Claude Fable/Mythos 5 reject forced tool use + * outright ("tool_choice forces tool use is not compatible with this model"); + * the request builder downgrades forced choices to `auto` when this is false. + * When unset, auto-detected from the model id. Default: true. + */ + supportsForcedToolChoice?: boolean; } /** From abc98e7711cbaf1f3b9c2bf2edba148cdd8bc6da Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:48:49 +0200 Subject: [PATCH 25/29] feat(packages/ai): added Anthropic catalog seeds and router fixes - Seeded Anthropic curated fallback models into generation before discovery sources. - Added claude-fable-5 and claude-mythos-5 model entries with 1,000,000 context and 128k max tokens. - Enabled reasoning, text,image input, thinking controls, and larger limits across many models. - Fixed OpenRouter Anthropic tool-call behavior for empty cache_control payloads. --- packages/ai/CHANGELOG.md | 2 + packages/ai/scripts/generate-models.ts | 6 + packages/ai/src/models.json | 964 ++++++++++-------- .../ai/src/provider-models/openai-compat.ts | 39 + 4 files changed, 605 insertions(+), 406 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 023fcbf32..1005ee80d 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,6 +5,8 @@ ### Fixed - Fixed `google-antigravity` not rotating to another stored OAuth account when Cloud Code Assist returns `429 You have exhausted your capacity on this model. Your quota will reset after …`. `parseRateLimitReason` matched the literal `capacity` before the `quota will reset` suffix and downgraded the failure to `MODEL_CAPACITY_EXHAUSTED` (45–75 s backoff), and `isUsageLimitError` returned false for the same message — so `markUsageLimitReached` was never invoked and the agent kept hammering the exhausted credential while the retry layer bailed on the multi-hour `retry-after`. Both paths now treat the Antigravity phrasing as `QUOTA_EXHAUSTED` / usage-limit, blocking the current credential until reset and letting the session pick an unblocked sibling ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). +- Fixed OpenRouter Anthropic chat-completions requests placing `cache_control` on empty assistant tool-call content. The cache marker now skips empty text and attaches to the most recent non-empty text part, avoiding HTTP 400 payloads with `{type:"text", text:"", cache_control:...}`. +- Fixed Fable-only Anthropic request shaping to cover Claude Mythos 5, and added Mythos 5 to the first-party Anthropic catalog seed. Adaptive display, sampling suppression, mid-conversation system messages, forced-tool-choice downgrade, and Bedrock adaptive metadata now handle both model families. ### Added diff --git a/packages/ai/scripts/generate-models.ts b/packages/ai/scripts/generate-models.ts index 144d3dade..9fb9a417d 100644 --- a/packages/ai/scripts/generate-models.ts +++ b/packages/ai/scripts/generate-models.ts @@ -27,6 +27,7 @@ import { PROVIDER_DESCRIPTORS, } from "../src/provider-models/descriptors"; import { + ANTHROPIC_CURATED_FALLBACK_MODELS, buildXaiOAuthStaticSeed, clampFireworksKimiMaxTokens, isFireworksKimiK2ModelId, @@ -382,6 +383,11 @@ async function generateModels() { // persisted `modelRoles.default = "xai-oauth/"` is honored before the // async refresh fires (interactive boot does not await refresh). allModels.push(...buildXaiOAuthStaticSeed()); + // Seed Anthropic models that are live on the first-party API or in limited + // release but that models.dev has not catalogued yet (e.g. Claude Fable 5 / + // Mythos 5). Deduped behind upstream entries; metadata is pinned in + // applyAnthropicCatalogPolicy. + allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS); const specialDiscoverySources = [ { label: "Antigravity", fetch: fetchAntigravityModels }, diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index 06454751b..63854d76d 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -4765,13 +4765,14 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "minimax/minimax-m3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -4779,8 +4780,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 512000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "minimax/MiniMax-Text-01": { "id": "minimax/MiniMax-Text-01", @@ -6432,13 +6438,14 @@ }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", - "name": "stepfun/step-3.7-flash", + "name": "Step 3.7 Flash", "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -6446,8 +6453,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 256000, + "maxTokens": 256000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "triposr": { "id": "triposr", @@ -8061,7 +8073,7 @@ "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -8073,7 +8085,12 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 4096 + "maxTokens": 4096, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } }, "global.anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "global.anthropic.claude-haiku-4-5-20251001-v1:0", @@ -8935,7 +8952,7 @@ "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -8946,7 +8963,12 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 16384 + "maxTokens": 16384, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } }, "openai.gpt-oss-120b-1:0": { "id": "openai.gpt-oss-120b-1:0", @@ -8954,7 +8976,7 @@ "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -8965,7 +8987,12 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 16384 + "maxTokens": 16384, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } }, "openai.gpt-oss-20b": { "id": "openai.gpt-oss-20b", @@ -8973,7 +9000,7 @@ "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -8984,7 +9011,12 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 16384 + "maxTokens": 16384, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } }, "openai.gpt-oss-20b-1:0": { "id": "openai.gpt-oss-20b-1:0", @@ -8992,7 +9024,7 @@ "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -9003,7 +9035,12 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 16384 + "maxTokens": 16384, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } }, "openai.gpt-oss-safeguard-120b": { "id": "openai.gpt-oss-safeguard-120b", @@ -9884,6 +9921,31 @@ "contextWindow": 200000, "maxTokens": 4096 }, + "claude-fable-5": { + "id": "claude-fable-5", + "name": "Claude Fable 5", + "api": "anthropic-messages", + "provider": "anthropic", + "baseUrl": "https://api.anthropic.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 50, + "cacheRead": 1, + "cacheWrite": 12.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", @@ -9934,6 +9996,31 @@ "maxLevel": "xhigh" } }, + "claude-mythos-5": { + "id": "claude-mythos-5", + "name": "Claude Mythos 5", + "api": "anthropic-messages", + "provider": "anthropic", + "baseUrl": "https://api.anthropic.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 50, + "cacheRead": 1, + "cacheWrite": 12.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "claude-opus-4-0": { "id": "claude-opus-4-0", "name": "Claude Opus 4 (latest)", @@ -15204,71 +15291,6 @@ } }, "google-vertex": { - "claude-3-5-haiku@20241022": { - "id": "claude-3-5-haiku@20241022", - "name": "Claude Haiku 3.5", - "api": "anthropic-messages", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-3-5-haiku@20241022:streamRawPredict", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.8, - "output": 4, - "cacheRead": 0.08, - "cacheWrite": 1 - }, - "contextWindow": 200000, - "maxTokens": 8192 - }, - "claude-3-5-sonnet@20241022": { - "id": "claude-3-5-sonnet@20241022", - "name": "Claude Sonnet 3.5 v2", - "api": "anthropic-messages", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-3-5-sonnet@20241022:streamRawPredict", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, - "cacheWrite": 3.75 - }, - "contextWindow": 200000, - "maxTokens": 8192 - }, - "claude-3-7-sonnet@20250219": { - "id": "claude-3-7-sonnet@20250219", - "name": "Claude Sonnet 3.7", - "api": "anthropic-messages", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-3-7-sonnet@20250219:streamRawPredict", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, - "cacheWrite": 3.75 - }, - "contextWindow": 200000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "minLevel": "minimal", - "maxLevel": "xhigh" - } - }, "claude-haiku-4-5@20251001": { "id": "claude-haiku-4-5@20251001", "name": "Claude Haiku 4.5", @@ -15294,31 +15316,6 @@ "maxLevel": "xhigh" } }, - "claude-opus-4-1@20250805": { - "id": "claude-opus-4-1@20250805", - "name": "Claude Opus 4.1", - "api": "anthropic-messages", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-opus-4-1@20250805:streamRawPredict", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 15, - "output": 75, - "cacheRead": 1.5, - "cacheWrite": 18.75 - }, - "contextWindow": 200000, - "maxTokens": 32000, - "thinking": { - "mode": "budget", - "minLevel": "minimal", - "maxLevel": "xhigh" - } - }, "claude-opus-4-5@20251101": { "id": "claude-opus-4-5@20251101", "name": "Claude Opus 4.5", @@ -15419,31 +15416,6 @@ "maxLevel": "xhigh" } }, - "claude-opus-4@20250514": { - "id": "claude-opus-4@20250514", - "name": "Claude Opus 4", - "api": "anthropic-messages", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-opus-4@20250514:streamRawPredict", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 15, - "output": 75, - "cacheRead": 1.5, - "cacheWrite": 18.75 - }, - "contextWindow": 200000, - "maxTokens": 32000, - "thinking": { - "mode": "budget", - "minLevel": "minimal", - "maxLevel": "xhigh" - } - }, "claude-sonnet-4-5@20250929": { "id": "claude-sonnet-4-5@20250929", "name": "Claude Sonnet 4.5", @@ -15494,31 +15466,6 @@ "maxLevel": "high" } }, - "claude-sonnet-4@20250514": { - "id": "claude-sonnet-4@20250514", - "name": "Claude Sonnet 4", - "api": "anthropic-messages", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-sonnet-4@20250514:streamRawPredict", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, - "cacheWrite": 3.75 - }, - "contextWindow": 200000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "minLevel": "minimal", - "maxLevel": "xhigh" - } - }, "deepseek-ai/deepseek-v3.1-maas": { "id": "deepseek-ai/deepseek-v3.1-maas", "name": "DeepSeek V3.1", @@ -15567,46 +15514,6 @@ "maxLevel": "xhigh" } }, - "gemini-2.0-flash": { - "id": "gemini-2.0-flash", - "name": "Gemini 2.0 Flash", - "api": "google-vertex", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.15, - "output": 0.6, - "cacheRead": 0.025, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 8192 - }, - "gemini-2.0-flash-lite": { - "id": "gemini-2.0-flash-lite", - "name": "Gemini 2.0 Flash-Lite", - "api": "google-vertex", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.075, - "output": 0.3, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 8192 - }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", @@ -15657,56 +15564,6 @@ "maxLevel": "high" } }, - "gemini-2.5-flash-lite-preview-06-17": { - "id": "gemini-2.5-flash-lite-preview-06-17", - "name": "Gemini 2.5 Flash Lite Preview 06-17", - "api": "google-vertex", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.1, - "output": 0.4, - "cacheRead": 0.025, - "cacheWrite": 0 - }, - "contextWindow": 65536, - "maxTokens": 65536, - "thinking": { - "mode": "budget", - "minLevel": "minimal", - "maxLevel": "high" - } - }, - "gemini-2.5-flash-preview-09-2025": { - "id": "gemini-2.5-flash-preview-09-2025", - "name": "Gemini 2.5 Flash Preview 09-25", - "api": "google-vertex", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.3, - "output": 2.5, - "cacheRead": 0.075, - "cacheWrite": 0.383 - }, - "contextWindow": 1048576, - "maxTokens": 65536, - "thinking": { - "mode": "budget", - "minLevel": "minimal", - "maxLevel": "high" - } - }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", @@ -15757,35 +15614,6 @@ "maxLevel": "high" } }, - "gemini-3-pro-preview": { - "id": "gemini-3-pro-preview", - "name": "Gemini 3 Pro Preview", - "api": "google-vertex", - "provider": "google-vertex", - "baseUrl": "https://{location}-aiplatform.googleapis.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2, - "output": 12, - "cacheRead": 0.2, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 65536, - "thinking": { - "mode": "google-level", - "minLevel": "low", - "maxLevel": "high", - "levels": [ - "low", - "high" - ] - } - }, "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", @@ -17283,6 +17111,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "anthropic/claude-fable-5": { + "id": "anthropic/claude-fable-5", + "name": "Anthropic: Claude Fable 5 ($$$$)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5", @@ -17468,13 +17315,14 @@ }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", - "name": "Anthropic: Claude Opus 4.8", + "name": "Claude Opus 4.8", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -17482,8 +17330,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "anthropic/claude-opus-4.8-fast": { "id": "anthropic/claude-opus-4.8-fast", @@ -18837,13 +18690,14 @@ }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", - "name": "Google: Gemini 3.1 Flash Lite", + "name": "Gemini 3.1 Flash Lite", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -18851,8 +18705,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", @@ -18924,13 +18783,14 @@ }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", - "name": "Google: Gemini 3.5 Flash", + "name": "Gemini 3.5 Flash", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -18938,8 +18798,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "google/gemma-2-27b-it": { "id": "google/gemma-2-27b-it", @@ -19340,7 +19205,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -19350,8 +19215,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 262000, + "maxTokens": 65000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "inclusionai/ring-2.6-1t:free": { "id": "inclusionai/ring-2.6-1t:free", @@ -20217,13 +20087,14 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax: MiniMax M3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -20231,8 +20102,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 512000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "minimax/minimax-m3:discounted": { "id": "minimax/minimax-m3:discounted", @@ -24236,11 +24112,11 @@ }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", - "name": "Qwen: Qwen3.7 Max", + "name": "Qwen3.7 Max", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -24250,18 +24126,24 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "qwen/qwen3.7-plus": { "id": "qwen/qwen3.7-plus", - "name": "Qwen: Qwen3.7 Plus", + "name": "Qwen3.7 Plus", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -24269,8 +24151,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "qwen/qwen3.7-plus:free": { "id": "qwen/qwen3.7-plus:free", @@ -24654,13 +24541,14 @@ }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", - "name": "StepFun: Step 3.7 Flash", + "name": "Step 3.7 Flash", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -24668,8 +24556,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 256000, + "maxTokens": 256000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "stepfun/step-3.7-flash:free": { "id": "stepfun/step-3.7-flash:free", @@ -25152,13 +25045,14 @@ }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", - "name": "xAI: Grok 4.3", + "name": "Grok 4.3", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -25166,18 +25060,24 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1000000, + "maxTokens": 1000000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", - "name": "xAI: Grok Build 0.1", + "name": "Grok Build 0.1", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -25185,8 +25085,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 256000, + "maxTokens": 256000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "x-ai/grok-code-fast-1": { "id": "x-ai/grok-code-fast-1", @@ -27754,6 +27659,31 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "anthropic/claude-fable-5": { + "id": "anthropic/claude-fable-5", + "name": "Anthropic: Claude Fable 5", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 50, + "cacheRead": 1, + "cacheWrite": 12.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "anthropic/claude-haiku-latest": { "id": "anthropic/claude-haiku-latest", "name": "anthropic/claude-haiku-latest", @@ -27825,13 +27755,14 @@ }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", - "name": "anthropic/claude-opus-4.8", + "name": "Claude Opus 4.8", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -27839,8 +27770,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888, + "contextWindow": 1000000, + "maxTokens": 128000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -31838,7 +31769,7 @@ }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", - "name": "Google: Gemini 3.1 Flash Lite", + "name": "Gemini 3.1 Flash Lite", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -31979,13 +31910,14 @@ }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", - "name": "google/gemini-3.5-flash", + "name": "Gemini 3.5 Flash", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -31993,8 +31925,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "google/gemini-3.5-flash-thinking": { "id": "google/gemini-3.5-flash-thinking", @@ -32509,11 +32446,11 @@ }, "inclusionai/ring-2.6-1t": { "id": "inclusionai/ring-2.6-1t", - "name": "inclusionai/ring-2.6-1t", + "name": "inclusionAI: Ring-2.6-1T", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -32523,8 +32460,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 262000, + "maxTokens": 65000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "Infermatic/MN-12B-Inferor-v0.0": { "id": "Infermatic/MN-12B-Inferor-v0.0", @@ -34303,13 +34245,14 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "minimax/minimax-m3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -34317,8 +34260,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888, + "contextWindow": 512000, + "maxTokens": 128000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -39263,6 +39206,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "TEE/deepseek-v4-flash": { + "id": "TEE/deepseek-v4-flash", + "name": "TEE/deepseek-v4-flash", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "TEE/deepseek-v4-pro": { "id": "TEE/deepseek-v4-pro", "name": "TEE/deepseek-v4-pro", @@ -39862,6 +39824,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "TEE/qwen3.6-35b-a3b": { + "id": "TEE/qwen3.6-35b-a3b", + "name": "TEE/qwen3.6-35b-a3b", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "TEE/qwen3.6-35b-a3b-uncensored": { "id": "TEE/qwen3.6-35b-a3b-uncensored", "name": "TEE/qwen3.6-35b-a3b-uncensored", @@ -40750,13 +40731,14 @@ }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", - "name": "x-ai/grok-4.3", + "name": "Grok 4.3", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -40764,18 +40746,24 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1000000, + "maxTokens": 1000000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", - "name": "x-ai/grok-build-0.1", + "name": "Grok Build 0.1", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -40783,8 +40771,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 256000, + "maxTokens": 256000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "x-ai/grok-code-fast-1": { "id": "x-ai/grok-code-fast-1", @@ -48252,6 +48245,30 @@ "maxLevel": "xhigh" } }, + "north-mini-code-free": { + "id": "north-mini-code-free", + "name": "North Mini Code Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", @@ -48859,6 +48876,31 @@ "maxLevel": "high" } }, + "anthropic/claude-fable-5": { + "id": "anthropic/claude-fable-5", + "name": "Anthropic: Claude Fable 5", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 50, + "cacheRead": 1, + "cacheWrite": 12.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5", @@ -49056,7 +49098,7 @@ }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", - "name": "Anthropic: Claude Opus 4.8", + "name": "Claude Opus 4.8", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -50140,7 +50182,7 @@ }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", - "name": "Google: Gemini 3.1 Flash Lite", + "name": "Gemini 3.1 Flash Lite", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -50243,7 +50285,7 @@ }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", - "name": "Google: Gemini 3.5 Flash", + "name": "Gemini 3.5 Flash", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -50931,8 +50973,8 @@ ], "cost": { "input": 0.15, - "output": 1.15, - "cacheRead": 0.03, + "output": 0.8999999999999999, + "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 204800, @@ -50981,13 +51023,13 @@ "text" ], "cost": { - "input": 0.27899999999999997, - "output": 1.2, - "cacheRead": 0.059, + "input": 0.27, + "output": 1.08, + "cacheRead": 0.054, "cacheWrite": 0 }, "contextWindow": 204800, - "maxTokens": 196608, + "maxTokens": 131072, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -50996,7 +51038,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax: MiniMax M3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -54862,7 +54904,7 @@ }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", - "name": "Qwen: Qwen3.7 Max", + "name": "Qwen3.7 Max", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -54886,7 +54928,7 @@ }, "qwen/qwen3.7-plus": { "id": "qwen/qwen3.7-plus", - "name": "Qwen: Qwen3.7 Plus", + "name": "Qwen3.7 Plus", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -55081,7 +55123,7 @@ }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", - "name": "StepFun: Step 3.7 Flash", + "name": "Step 3.7 Flash", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -55502,7 +55544,7 @@ }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", - "name": "xAI: Grok 4.3", + "name": "Grok 4.3", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -55518,7 +55560,7 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 8888, + "maxTokens": 1000000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -55527,7 +55569,7 @@ }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", - "name": "xAI: Grok Build 0.1", + "name": "Grok Build 0.1", "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -55543,7 +55585,7 @@ "cacheWrite": 0 }, "contextWindow": 256000, - "maxTokens": 8888, + "maxTokens": 256000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -56823,6 +56865,34 @@ "supportsUsageInStreaming": false } }, + "claude-fable-5": { + "id": "claude-fable-5", + "name": "Claude Fable 5", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", @@ -58977,6 +59047,28 @@ "supportsUsageInStreaming": false } }, + "tencent-hy3-preview": { + "id": "tencent-hy3-preview", + "name": "tencent-hy3-preview", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 8888, + "compat": { + "supportsUsageInStreaming": false + } + }, "venice-uncensored": { "id": "venice-uncensored", "name": "Venice Uncensored 1.1", @@ -59836,6 +59928,31 @@ "maxLevel": "xhigh" } }, + "anthropic/claude-fable-5": { + "id": "anthropic/claude-fable-5", + "name": "Claude Fable 5", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 50, + "cacheRead": 1, + "cacheWrite": 12.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5", @@ -61136,7 +61253,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax M3", + "name": "MiniMax-M3", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -64225,7 +64342,7 @@ "cacheRead": 0.2, "cacheWrite": 0 }, - "contextWindow": 2000000, + "contextWindow": 1000000, "maxTokens": 30000 }, "grok-4.20-0309-reasoning": { @@ -64245,7 +64362,7 @@ "cacheRead": 0.2, "cacheWrite": 0 }, - "contextWindow": 2000000, + "contextWindow": 1000000, "maxTokens": 30000, "thinking": { "mode": "effort", @@ -65273,7 +65390,7 @@ }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", - "name": "Anthropic: Claude Opus 4.8", + "name": "Claude Opus 4.8", "api": "anthropic-messages", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/anthropic", @@ -65286,10 +65403,10 @@ "input": 5, "output": 25, "cacheRead": 0.5, - "cacheWrite": 10 + "cacheWrite": 6.25 }, "contextWindow": 1000000, - "maxTokens": 8888, + "maxTokens": 128000, "thinking": { "mode": "anthropic-adaptive", "minLevel": "minimal", @@ -66019,7 +66136,7 @@ }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", - "name": "Google: Gemini 3.1 Flash Lite", + "name": "Gemini 3.1 Flash Lite", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -66035,7 +66152,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 8888, + "maxTokens": 65536, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -66093,7 +66210,7 @@ }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", - "name": "Google: Gemini 3.5 Flash", + "name": "Gemini 3.5 Flash", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -66109,7 +66226,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 8888, + "maxTokens": 65536, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -66354,8 +66471,8 @@ "cacheRead": 0.06, "cacheWrite": 0 }, - "contextWindow": 262144, - "maxTokens": 8888, + "contextWindow": 262000, + "maxTokens": 65000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -66671,7 +66788,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax: MiniMax M3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -66681,13 +66798,13 @@ "image" ], "cost": { - "input": 0.3, - "output": 1.2, - "cacheRead": 0.06, + "input": 0.6, + "output": 2.4, + "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 512000, - "maxTokens": 8888, + "maxTokens": 128000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -67466,6 +67583,31 @@ "maxLevel": "xhigh" } }, + "openai/gpt-5.5-instant": { + "id": "openai/gpt-5.5-instant", + "name": "GPT-5.5 Instant", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 400000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "low", + "maxLevel": "xhigh" + } + }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", @@ -67892,7 +68034,7 @@ }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", - "name": "Qwen: Qwen3.7-Max", + "name": "Qwen3.7 Max", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -67901,13 +68043,13 @@ "text" ], "cost": { - "input": 1.25, - "output": 3.75, - "cacheRead": 0.125, - "cacheWrite": 1.5625 + "input": 2.5, + "output": 7.5, + "cacheRead": 0.5, + "cacheWrite": 3.125 }, "contextWindow": 1000000, - "maxTokens": 8888, + "maxTokens": 65536, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -67916,7 +68058,7 @@ }, "qwen/qwen3.7-plus": { "id": "qwen/qwen3.7-plus", - "name": "Qwen: Qwen3.7-Plus", + "name": "Qwen3.7 Plus", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -67928,11 +68070,11 @@ "cost": { "input": 0.4, "output": 1.6, - "cacheRead": 0.04, + "cacheRead": 0.08, "cacheWrite": 0.5 }, "contextWindow": 1000000, - "maxTokens": 8888, + "maxTokens": 64000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -68103,7 +68245,7 @@ }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", - "name": "StepFun: Step 3.7 Flash", + "name": "Step 3.7 Flash", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -68115,11 +68257,11 @@ "cost": { "input": 0.2, "output": 1.15, - "cacheRead": 0.04, + "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 256000, - "maxTokens": 8888, + "maxTokens": 256000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -68505,11 +68647,11 @@ }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", - "name": "xAI: Grok 4.3", + "name": "Grok 4.3", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -68521,15 +68663,20 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 8888 + "maxTokens": 1000000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", - "name": "xAI: Grok Build 0.1", + "name": "Grok Build 0.1", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -68541,7 +68688,12 @@ "cacheWrite": 0 }, "contextWindow": 256000, - "maxTokens": 8888 + "maxTokens": 256000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "x-ai/grok-code-fast-1": { "id": "x-ai/grok-code-fast-1", diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index ad99ca1df..a7ac02edb 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -145,6 +145,43 @@ function buildAnthropicReferenceMap( return merged; } +/** + * Curated Anthropic models that are live or limited-availability on the + * first-party `/v1/models` endpoint but that models.dev has not catalogued yet. + * Seeded into model generation so the bundled catalog is never gated on + * models.dev's update cadence; deduped behind upstream catalog / models.dev + * entries once those appear. Token limits and pricing are pinned + * authoritatively in + * `applyAnthropicCatalogPolicy`, and `thinking` is derived by + * `refreshModelThinking` during generation. + */ +export const ANTHROPIC_CURATED_FALLBACK_MODELS: readonly Model<"anthropic-messages">[] = [ + { + id: "claude-fable-5", + name: "Claude Fable 5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text", "image"], + cost: { input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }, + contextWindow: 1_000_000, + maxTokens: 128_000, + }, + { + id: "claude-mythos-5", + name: "Claude Mythos 5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text", "image"], + cost: { input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }, + contextWindow: 1_000_000, + maxTokens: 128_000, + }, +]; + function mapWithBundledReference( entry: OpenAICompatibleModelRecord, defaults: Model, @@ -2615,6 +2652,8 @@ export function mapModelsDevToModels( // Bedrock cross-region prefix helpers const BEDROCK_GLOBAL_PREFIXES = [ + "anthropic.claude-fable-5", + "anthropic.claude-mythos-5", "anthropic.claude-haiku-4-5", "anthropic.claude-sonnet-4", "anthropic.claude-opus-4-5", From ba5b9ed617399d2f4f63a89fe620820456cb28f2 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:48:49 +0200 Subject: [PATCH 26/29] test(packages/ai): added Mythos and cache-control regression coverage - Added Anthropic payload assertions for Claude Fable/Mythos 5 in alignment tests. - Added role-to-system mapping test coverage for Claude Mythos 5 conversation messages. - Added issue-1373 regression checks for Mythos default adaptive thinking on Bedrock. - Added model policy expectations for Claude Mythos 5 generated metadata and effort mapping. - Added cache-control regression test for empty assistant tool-call content in OpenAI completions. --- packages/ai/test/anthropic-alignment.test.ts | 74 +++++++++++++++++++ .../anthropic-mid-conversation-system.test.ts | 16 ++-- packages/ai/test/issue-1373-repro.test.ts | 12 +++ packages/ai/test/model-thinking.test.ts | 47 ++++++++++++ .../ai/test/openai-completions-compat.test.ts | 60 ++++++++++++++- 5 files changed, 203 insertions(+), 6 deletions(-) diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index d6346815e..a2888c6b3 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -1439,6 +1439,37 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.thinking).toBeUndefined(); }); + it("drops sampling params for Claude Fable/Mythos 5 without enabled thinking", async () => { + for (const id of ["claude-fable-5", "claude-mythos-5"] as const) { + const payload = (await captureAnthropicPayload( + { + ...ANTHROPIC_MODEL, + id, + name: id === "claude-fable-5" ? "Claude Fable 5" : "Claude Mythos 5", + }, + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], + }, + { + temperature: 0.2, + topP: 0.3, + topK: 4, + }, + )) as { + temperature?: number; + top_p?: number; + top_k?: number; + thinking?: { type?: string }; + }; + + expect(payload.temperature).toBeUndefined(); + expect(payload.top_p).toBeUndefined(); + expect(payload.top_k).toBeUndefined(); + expect(payload.thinking).toBeUndefined(); + } + }); + it("drops sampling params and keeps summarized adaptive thinking for OAuth Opus 4.7+", async () => { const payload = (await captureAnthropicPayload( { @@ -1616,6 +1647,49 @@ describe("Anthropic request fingerprint alignment", () => { }); }); + it("downgrades forced tool choice for Claude Fable/Mythos without deleting adaptive thinking", async () => { + for (const id of ["claude-fable-5", "claude-mythos-5"] as const) { + const payload = (await captureAnthropicPayload( + { + ...ANTHROPIC_MODEL, + id, + name: id === "claude-fable-5" ? "Claude Fable 5" : "Claude Mythos 5", + contextWindow: 1_000_000, + maxTokens: 128_000, + thinking: { + mode: "anthropic-adaptive", + minLevel: Effort.Minimal, + maxLevel: Effort.XHigh, + }, + }, + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Use the tool", timestamp: Date.now() }], + tools: [ + { + name: "lookup", + description: "Lookup a value", + parameters: { type: "object", properties: {}, additionalProperties: false }, + }, + ], + }, + { + thinkingEnabled: true, + reasoning: Effort.High, + toolChoice: "any", + }, + )) as { + thinking?: { type?: string; display?: string }; + tool_choice?: { type?: string }; + output_config?: { effort?: string }; + }; + + expect(payload.tool_choice).toEqual({ type: "auto" }); + expect(payload.thinking).toEqual({ type: "adaptive", display: "summarized" }); + expect(payload.output_config).toEqual({ effort: "xhigh" }); + } + }); + it("treats tool prefix helpers as no-ops when prefix is empty string", () => { // Directly verify the codec's identity behaviour: builtins pass through apply unchanged. // (Empty-prefix path is exercised by the builtin guard below; the contract is diff --git a/packages/ai/test/anthropic-mid-conversation-system.test.ts b/packages/ai/test/anthropic-mid-conversation-system.test.ts index 198291f9a..26dcc9e00 100644 --- a/packages/ai/test/anthropic-mid-conversation-system.test.ts +++ b/packages/ai/test/anthropic-mid-conversation-system.test.ts @@ -3,11 +3,11 @@ import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { AssistantMessage, DeveloperMessage, Message, Model, UserMessage } from "@oh-my-pi/pi-ai/types"; /** - * Claude Opus 4.8 introduced mid-conversation `role: "system"` messages. Our - * `developer` messages (the system-priority instructions we already emit as - * `developer`/`system` to OpenAI providers) should map to that role on models - * that support it, while respecting Anthropic's placement rules and falling - * back to `user` everywhere else. + * Claude Opus 4.8 and the Fable/Mythos 5 generation support mid-conversation + * `role: "system"` messages. Our `developer` messages (the system-priority + * instructions we already emit as `developer`/`system` to OpenAI providers) + * should map to that role on models that support it, while respecting + * Anthropic's placement rules and falling back to `user` everywhere else. * @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages */ @@ -73,6 +73,12 @@ describe("Anthropic mid-conversation system messages", () => { expect(params.at(-1)?.role).toBe("system"); }); + it("maps developer messages to system on Claude Mythos 5", () => { + const model = makeModel({ id: "claude-mythos-5", name: "Claude Mythos 5" }); + const params = convertAnthropicMessages([user("hi"), developer("Use project rules.")], model, false); + expect(params.map(p => p.role)).toEqual(["user", "system"]); + }); + it("maps a developer message that precedes an assistant turn to role: system", () => { const model = makeModel(); const params = convertAnthropicMessages( diff --git a/packages/ai/test/issue-1373-repro.test.ts b/packages/ai/test/issue-1373-repro.test.ts index ce0e65111..5173d803f 100644 --- a/packages/ai/test/issue-1373-repro.test.ts +++ b/packages/ai/test/issue-1373-repro.test.ts @@ -90,6 +90,18 @@ describe("issue #1373: Bedrock Claude thinkingDisplay", () => { }); }); + it("defaults adaptive thinking to display=summarized on Fable/Mythos 5", async () => { + for (const id of ["global.anthropic.claude-fable-5", "global.anthropic.claude-mythos-5"] as const) { + const payload = await captureBedrockPayload(adaptiveModel(id), { + reasoning: Effort.High, + }); + expect(payload.additionalModelRequestFields?.thinking).toMatchObject({ + type: "adaptive", + display: "summarized", + }); + } + }); + it("respects explicit thinkingDisplay='omitted' on Opus 4.7+", async () => { const payload = await captureBedrockPayload(adaptiveModel("eu.anthropic.claude-opus-4-7"), { reasoning: Effort.High, diff --git a/packages/ai/test/model-thinking.test.ts b/packages/ai/test/model-thinking.test.ts index e8475974c..69dadd0b0 100644 --- a/packages/ai/test/model-thinking.test.ts +++ b/packages/ai/test/model-thinking.test.ts @@ -107,6 +107,16 @@ describe("model thinking metadata", () => { api: "anthropic-messages", provider: "anthropic", }); + const mythos = createModel({ + id: "claude-mythos-5", + api: "anthropic-messages", + provider: "anthropic", + }); + const mythosBedrock = createModel({ + id: "global.anthropic.claude-mythos-5", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + }); expect(opus45.thinking?.mode).toBe("anthropic-budget-effort"); expect(opus46.thinking?.mode).toBe("anthropic-adaptive"); @@ -121,6 +131,12 @@ describe("model thinking metadata", () => { minLevel: Effort.Minimal, maxLevel: Effort.High, }); + expect(mythos.thinking).toEqual({ + mode: "anthropic-adaptive", + minLevel: Effort.Minimal, + maxLevel: Effort.XHigh, + }); + expect(mythosBedrock.thinking?.mode).toBe("anthropic-adaptive"); // Opus 4.6 has no real xhigh level — pi-ai aliases XHigh to Anthropic's "max". expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toBe("max"); // Opus 4.7+ on the Messages API exposes the full five-tier scale, so pi-ai @@ -130,6 +146,9 @@ describe("model thinking metadata", () => { expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Medium)).toBe("high"); expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("xhigh"); expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max"); + expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh"); + expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.XHigh)).toBe("max"); + expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max"); // Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max". expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high"); expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.XHigh)).toBe("max"); @@ -218,6 +237,34 @@ describe("generated model policies", () => { expect(models[3]?.priority).toBe(1); }); + it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => { + const models: Model[] = [ + { + id: "claude-mythos-5", + name: "Claude Mythos 5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://example.com", + reasoning: true, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200000, + maxTokens: 32000, + }, + ]; + + applyGeneratedModelPolicies(models); + + expect(models[0]?.contextWindow).toBe(1_000_000); + expect(models[0]?.maxTokens).toBe(128_000); + expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }); + expect(models[0]?.thinking).toEqual({ + mode: "anthropic-adaptive", + minLevel: Effort.Minimal, + maxLevel: Effort.XHigh, + }); + }); + it("normalizes Copilot generated fallback limits", () => { const models: Model[] = [ { diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 6d499fa7a..7bab53c0d 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -7,7 +7,14 @@ import { streamOpenAICompletions, } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { type ResolvedOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-ai/providers/openai-completions-compat"; -import type { AssistantMessage, Context, FetchImpl, Model, OpenAICompat } from "@oh-my-pi/pi-ai/types"; +import type { + AssistantMessage, + Context, + FetchImpl, + Model, + OpenAICompat, + ToolResultMessage, +} from "@oh-my-pi/pi-ai/types"; function createAbortedSignal(): AbortSignal { const controller = new AbortController(); @@ -1469,6 +1476,57 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () => expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" }); }); + it("does not attach Anthropic cache_control to empty assistant tool-call content", async () => { + const model = getBundledModel("openrouter", "anthropic/claude-sonnet-4") as Model<"openai-completions">; + const assistantMessage: AssistantMessage = { + role: "assistant", + content: [ + { + type: "toolCall", + id: "call_read", + name: "read", + arguments: { path: "screenshot.png" }, + }, + ], + api: model.api, + provider: model.provider, + model: model.id, + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 1, + }; + const toolResultMessage: ToolResultMessage = { + role: "toolResult", + toolCallId: "call_read", + toolName: "read", + content: [{ type: "text", text: "Read image file [image/webp]" }], + isError: false, + timestamp: 2, + }; + const payload = await captureOpenAICompletionsPayload(model, { + messages: [{ role: "user", content: "cache me", timestamp: 0 }, assistantMessage, toolResultMessage], + }); + const messages = getPayloadMessages(payload); + const assistant = messages.find(message => { + const toolCalls = Reflect.get(message, "tool_calls"); + return Reflect.get(message, "role") === "assistant" && Array.isArray(toolCalls); + }); + const firstUser = messages.find(message => Reflect.get(message, "role") === "user"); + const userContent = firstUser ? Reflect.get(firstUser, "content") : undefined; + const textPart = getLastTextPart(userContent); + + expect(assistant ? Reflect.get(assistant, "content") : undefined).toBe(""); + expect(Reflect.get(textPart ?? {}, "text")).toBe("cache me"); + expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" }); + }); + it("does not infer Anthropic cache_control for custom Claude ids without compat", async () => { const payload = await captureOpenAICompletionsPayload(claudeProxyModel(), cacheContext()); From 8c1fe6815c19c88da58645976e6443893f29439c Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:49:01 +0200 Subject: [PATCH 27/29] docs(tui): documented ghostty inline image startup fix in tui changelog - Documented the Ghostty inline image startup fix in `packages/tui/CHANGELOG.md`. --- packages/tui/CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index d05ad2d29..5e864102d 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -6,6 +6,10 @@ - Added a `wrapDescription` option to `SelectListLayoutOptions`. When enabled, long descriptions wrap onto continuation rows indented under the description column instead of being silently truncated. The slash-command/skill autocomplete picker now opts in so descriptions like the bundled skills' remain fully readable at normal terminal widths. `maxVisible` becomes the picker's visual row budget so the popup height stays bounded even when items wrap (a single 5-row description with `maxVisible=3` clips with the scrollbar carrying the offscreen tail). Navigation stays item-to-item, the narrow-width fallback (`width <= 40`) is unchanged, and the `ScrollView` scrollbar tracks visual rows so the thumb stays correct when items wrap unevenly. ([#2169](https://github.com/can1357/oh-my-pi/issues/2169)) +### Fixed + +- Fixed Ghostty's first inline image in a fresh TUI session sometimes rendering as an empty placeholder block by holding the initial Kitty graphics paint until the terminal startup settle window has passed. Direct Kitty placements also keep their zero-width reservation rows non-plain so image-only transcript blocks do not collapse when blank-edge trimming runs. + ## [15.10.8] - 2026-06-09 ### Fixed From 8001ec27730c0d9508feac6184a54e9940332646 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:49:01 +0200 Subject: [PATCH 28/29] test(tests): replaced scratch repro with anthropic request-shaping tests - Added Anthropic request-shaping coverage for adaptive and non-adaptive models in packages/ai/test. - Removed obsolete scratch thinking test used for commit-boundary tracing in coding-agent. --- packages/ai/CHANGELOG.md | 16 +-- .../anthropic-fable-request-shaping.test.ts | 99 +++++++++++++++++++ packages/coding-agent/CHANGELOG.md | 16 +-- .../modes/components/transcript-container.ts | 7 +- .../scratch-thinking-commit-repro.test.ts | 80 --------------- 5 files changed, 115 insertions(+), 103 deletions(-) create mode 100644 packages/ai/test/anthropic-fable-request-shaping.test.ts delete mode 100644 packages/coding-agent/test/scratch-thinking-commit-repro.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 1005ee80d..aee3d68d2 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,21 +2,21 @@ ## [Unreleased] +### Added + +- Added `antigravityRankingStrategy` and registered it as the default `CredentialRankingStrategy` for `google-antigravity`, so multi-account selection consumes the per-counter Antigravity usage reports (sorted ascending by `remainingFraction` in `fetchAntigravityUsage`) before falling back to round-robin — preventing the exhausted-counter credential from being chosen first when an unblocked sibling has headroom ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). +- Added Claude Fable 5 to the first-party Anthropic catalog, seeded directly via `ANTHROPIC_CURATED_FALLBACK_MODELS` rather than waiting on models.dev (1M context / 128k output, adaptive thinking, $10/$50 per MTok). The model parser recognizes the `fable` kind so effort tiers (low→max), adaptive thinking, and Opus-4.7-style sampling restrictions apply; token limits and pricing are pinned in `applyAnthropicCatalogPolicy`. + ### Fixed - Fixed `google-antigravity` not rotating to another stored OAuth account when Cloud Code Assist returns `429 You have exhausted your capacity on this model. Your quota will reset after …`. `parseRateLimitReason` matched the literal `capacity` before the `quota will reset` suffix and downgraded the failure to `MODEL_CAPACITY_EXHAUSTED` (45–75 s backoff), and `isUsageLimitError` returned false for the same message — so `markUsageLimitReached` was never invoked and the agent kept hammering the exhausted credential while the retry layer bailed on the multi-hour `retry-after`. Both paths now treat the Antigravity phrasing as `QUOTA_EXHAUSTED` / usage-limit, blocking the current credential until reset and letting the session pick an unblocked sibling ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). - Fixed OpenRouter Anthropic chat-completions requests placing `cache_control` on empty assistant tool-call content. The cache marker now skips empty text and attaches to the most recent non-empty text part, avoiding HTTP 400 payloads with `{type:"text", text:"", cache_control:...}`. - Fixed Fable-only Anthropic request shaping to cover Claude Mythos 5, and added Mythos 5 to the first-party Anthropic catalog seed. Adaptive display, sampling suppression, mid-conversation system messages, forced-tool-choice downgrade, and Bedrock adaptive metadata now handle both model families. - -### Added - -- Added `antigravityRankingStrategy` and registered it as the default `CredentialRankingStrategy` for `google-antigravity`, so multi-account selection consumes the per-counter Antigravity usage reports (sorted ascending by `remainingFraction` in `fetchAntigravityUsage`) before falling back to round-robin — preventing the exhausted-counter credential from being chosen first when an unblocked sibling has headroom ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). - -### Fixed - +- Fixed adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) returning HTTP 400 `"thinking.type.disabled" is not supported for this model` whenever thinking was turned off (utility calls and forced-tool turns route through the disable path). These models accept only `thinking.type: "adaptive"`; the request builder now omits the thinking field and pins the lowest adaptive effort instead of emitting `type: "disabled"`. - Widened the OpenAI-completions first-event watchdog floor from 120s to 300s for DeepSeek V4 reasoning models hosted on the official DeepSeek API. The reasoner emits no SSE bytes until its private chain-of-thought finishes, which routinely takes longer than the generic 100s first-event budget under load — every chat then aborted with `OpenAI completions stream timed out while waiting for the first event` and silently retried. Mirrors the existing GLM coding-plan widening ([#2177](https://github.com/can1357/oh-my-pi/issues/2177)). ## [15.10.8] - 2026-06-09 + ### Added - Added optional `fetch` transport override (`fetch?: FetchImpl`) to Google, Ollama, and OpenAI-compatible model-manager options so dynamic model discovery and metadata lookups can use a caller-supplied HTTP client instead of only global `fetch` @@ -3116,4 +3116,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_ ## [0.9.4] - 2025-11-26 -Initial release with multi-provider LLM support. \ No newline at end of file +Initial release with multi-provider LLM support. diff --git a/packages/ai/test/anthropic-fable-request-shaping.test.ts b/packages/ai/test/anthropic-fable-request-shaping.test.ts new file mode 100644 index 000000000..fa65fd833 --- /dev/null +++ b/packages/ai/test/anthropic-fable-request-shaping.test.ts @@ -0,0 +1,99 @@ +import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-ai/effort"; +import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; + +function makeAnthropicModel(id: string): Model<"anthropic-messages"> { + return { + id, + name: id, + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 128_000, + }; +} + +/** Adaptive-thinking model (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5). */ +function adaptiveModel(id: string): Model<"anthropic-messages"> { + return { + ...makeAnthropicModel(id), + thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh }, + }; +} + +const CONTEXT: Context = { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "weather in paris?", timestamp: Date.now() }], +}; + +function abortedSignal(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +type CapturedPayload = { + thinking?: { type: string }; + tool_choice?: { type: string }; + output_config?: { effort?: string }; +}; + +function capturePayload( + model: Model<"anthropic-messages">, + opts: Parameters[2], +): Promise { + const { promise, resolve } = Promise.withResolvers(); + streamAnthropic(model, CONTEXT, { + apiKey: "sk-ant-oat-test", + isOAuth: true, + signal: abortedSignal(), + onPayload: payload => resolve(payload as CapturedPayload), + ...opts, + }); + return promise; +} + +describe("Anthropic Fable/Mythos forced tool_choice", () => { + it("downgrades a forced tool to auto for Fable (which rejects forced tool use)", async () => { + const payload = await capturePayload(adaptiveModel("claude-fable-5"), { + toolChoice: { type: "tool", name: "get_weather" }, + }); + expect(payload.tool_choice?.type).toBe("auto"); + }); + + it("downgrades tool_choice:'any' to auto for Mythos", async () => { + const payload = await capturePayload(adaptiveModel("claude-mythos-5"), { + toolChoice: "any", + }); + expect(payload.tool_choice?.type).toBe("auto"); + }); + + it("preserves a forced tool_choice for non-Fable models (Opus 4.8 supports it)", async () => { + const payload = await capturePayload(adaptiveModel("claude-opus-4-8"), { + toolChoice: { type: "tool", name: "get_weather" }, + }); + expect(payload.tool_choice?.type).toBe("tool"); + }); +}); + +describe("Anthropic adaptive-only thinking disable", () => { + it("never sends thinking.type:'disabled' to an adaptive-only model, pins lowest effort", async () => { + const payload = await capturePayload(adaptiveModel("claude-fable-5"), { + thinkingEnabled: false, + }); + expect(payload.thinking).toBeUndefined(); + expect(payload.output_config?.effort).toBe("low"); + }); + + it("still sends thinking.type:'disabled' for budget-based (non-adaptive) models", async () => { + const payload = await capturePayload(makeAnthropicModel("claude-3-7-sonnet-20250219"), { + thinkingEnabled: false, + }); + expect(payload.thinking?.type).toBe("disabled"); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b2859b029..fd8a5927d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,24 +4,14 @@ ### Fixed +- Fixed streaming thinking (and other styled assistant content) vanishing from native scrollback once it scrolled past the viewport top during a foreground turn. The transcript's append-only commit detector compared raw row bytes, so a styled paragraph wrapping onto a new row (the span-closing SGR and width padding move while the visible cells stay identical) or a streamed token pushing the last word down a line flagged the block as permanently volatile — the commit boundary froze and every later row that crossed the viewport top was committed nowhere. Rows are now compared by visible content, a wrap-shrink of the in-flight bottom line counts as append-only, and a genuine one-off interior rewrite only suspends commits until the block re-earns append-only (30 clean frames), after which the pinned emitter backfills the stalled gap contiguously. Periodically rewriting blocks (spinners, collapsing tool previews) never re-earn and stay deferred. + - Fixed bracketed pastes containing multiple image file paths so each image is attached in order instead of treating the whole paste as one unreadable path. - Fixed MCP OAuth fallback prompts so the "Click here to authorize" label emits an auth-safe terminal hyperlink even when hyperlink auto-detection is unavailable, keeping non-browser MCP setup usable ([#2196](https://github.com/can1357/oh-my-pi/issues/2196)). - -### Fixed - - Fixed `task`-spawned subagents repeating filesystem scans the parent had already completed. `ExecutorOptions` and the `createAgentSession()` call inside `runSubprocess()` did not forward `rules`, the discovered extension paths, or the discovered `.omp/tools/` paths, so each subagent re-ran `loadCapability()`, `discoverAndLoadExtensions()`, and the full `.omp/tools/` walk. The toolsession now caches `session.rules`, `session.extensionPaths`, and `session.customToolPaths`; `runSubprocess()` threads them through; and `createAgentSession()` accepts new `preloadedExtensionPaths` and `preloadedCustomToolPaths` options backed by new exported `discoverExtensionPaths()` and `discoverCustomToolPaths()` helpers. Crucially, only path lists are forwarded — never loaded instances. Each session rebuilds its own `Extension` and `LoadedCustomTool` objects so the per-session `ExtensionAPI`/`CustomToolAPI` (cwd, eventBus, runtime, exec, pushPendingAction, UI) targets the right session; forwarding loaded instances would have routed extension handlers and custom-tool execution back through the parent. The CLI's `preloadedExtensions` short-circuit is preserved for same-process reuse and now shallow-clones the caller's `extensions` array so inline-extension augmentation (autoresearch + custom-tools wrapper) cannot bleed back into it ([#2190](https://github.com/can1357/oh-my-pi/issues/2190)). - -### Fixed - - Fixed SSH tool cancellation hanging behind OpenSSH ControlMaster streams that stayed open after an Esc/user interrupt ([#2180](https://github.com/can1357/oh-my-pi/issues/2180)). - -### Fixed - - Fixed Windows stdio MCP servers launched through PATH shims such as `codegraph.cmd` so bare commands like `codegraph` resolve via `PATHEXT` before spawn ([#2174](https://github.com/can1357/oh-my-pi/issues/2174)). - -### Fixed - - Fixed compiled-binary extensions failing to load `@oh-my-pi/pi-*` packages when `bun --compile` quietly dropped one of the extra entrypoints (observed on macOS arm64 release builds): the legacy-pi compat shim's package-root override branch returned the bunfs path without checking the target was present, so the rewrite emitted a `file://` URL to a missing module and the #1216 fallback (scoped to the throwing `getResolvedSpecifier` path) never ran. Override targets are now validated against the on-disk filesystem at module init, missing entries are dropped, and resolution falls through to canonical lookup so Bun resolves the import from the extension's own `node_modules` ([#2168](https://github.com/can1357/oh-my-pi/issues/2168)). ## [15.10.8] - 2026-06-09 @@ -9818,4 +9808,4 @@ Initial public release. - Git branch display in footer - Message queueing during streaming responses - OAuth integration for Gmail and Google Calendar access -- HTML export with syntax highlighting and collapsible sections \ No newline at end of file +- HTML export with syntax highlighting and collapsible sections diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index 5e6a36262..15ba42219 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -212,8 +212,11 @@ export class TranscriptContainer extends Container implements NativeScrollbackLi #nativeScrollbackLiveRegionStart: number | undefined; // Local line index up to which the leading run of live blocks is safe to // commit. Finalized blocks contribute their full frozen body; still-live - // blocks contribute only after their stripped render has been observed - // growing without changing a previously rendered interior row. + // blocks contribute only while their render has been observed growing + // without visibly rewriting a previously rendered interior row (escape + // placement and pad drift are ignored). A rewrite suspends the block's + // contribution until it re-earns append-only via VOLATILE_REARM_FRAMES + // clean frames; the pinned emitter then backfills the stalled gap. #nativeScrollbackCommitSafeEnd: number | undefined; override invalidate(): void { diff --git a/packages/coding-agent/test/scratch-thinking-commit-repro.test.ts b/packages/coding-agent/test/scratch-thinking-commit-repro.test.ts deleted file mode 100644 index 5689c41b9..000000000 --- a/packages/coding-agent/test/scratch-thinking-commit-repro.test.ts +++ /dev/null @@ -1,80 +0,0 @@ -import { beforeAll, describe, expect, it } from "bun:test"; -import type { AssistantMessage } from "@oh-my-pi/pi-ai"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message"; -import { TranscriptContainer } from "@oh-my-pi/pi-coding-agent/modes/components/transcript-container"; -import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { TERMINAL } from "@oh-my-pi/pi-tui"; - -type MutableTerminalInfo = { eagerEraseScrollbackRisk: boolean }; -const mutableTerminalInfo = TERMINAL as unknown as MutableTerminalInfo; - -function makeThinkingMessage(thinking: string): AssistantMessage { - return { - role: "assistant", - content: [{ type: "thinking", thinking }], - api: "anthropic", - provider: "anthropic", - model: "test-model", - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "stop", - timestamp: Date.now(), - }; -} - -const PARA = (n: number) => - `I'm realizing paragraph ${n} that thinking inference only happens when model.reasoning is enabled, so for a live-discovered model to get adaptive thinking configured, it needs reasoning true set upfront. That value comes from the defaults in the OpenAI-compatible models fetch, which gets applied when mapping a model without an existing reference.`; - -describe("scratch: streaming thinking commit boundary", () => { - beforeAll(async () => { - await initTheme(); - await Settings.init({ inMemory: true, cwd: process.cwd() }); - }); - - it("traces commitSafeEnd while streaming styled thinking", () => { - const saved = TERMINAL.eagerEraseScrollbackRisk; - mutableTerminalInfo.eagerEraseScrollbackRisk = true; - try { - const chat = new TranscriptContainer(); - const component = new AssistantMessageComponent(undefined, false); - chat.addChild(component); - - const fullText = [PARA(1), PARA(2), PARA(3)].join("\n\n"); - const words = fullText.split(" "); - - let prevRows: string[] = []; - let firstStall: number | undefined; - // stream ~5 words per frame (30fps coalescing) - for (let i = 5; i <= words.length; i += 5) { - const text = words.slice(0, i).join(" "); - component.updateContent(makeThinkingMessage(text)); - const rows = chat.render(120); - const safeEnd = chat.getNativeScrollbackCommitSafeEnd(); - if (safeEnd === undefined && rows.length > 3 && firstStall === undefined) { - firstStall = i; - // dump the frame diff that broke append-only detection - console.log(`--- stall at word ${i}, rows=${rows.length}, prevRows=${prevRows.length}`); - const limit = Math.max(prevRows.length, rows.length); - for (let r = 0; r < limit; r++) { - if (prevRows[r] !== rows[r]) { - console.log(` row ${r} prev: ${JSON.stringify(prevRows[r])}`); - console.log(` row ${r} cur : ${JSON.stringify(rows[r])}`); - } - } - } - prevRows = rows; - } - console.log(`total words=${words.length}, firstStall=${firstStall}`); - expect(firstStall).toBeUndefined(); - } finally { - mutableTerminalInfo.eagerEraseScrollbackRisk = saved; - } - }); -}); From 7207929c72c4eee66f3f0229c3315b047391f6cc Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 19:52:12 +0200 Subject: [PATCH 29/29] chore: bump version to 15.10.9 --- Cargo.lock | 20 +++++++------- Cargo.toml | 2 +- bun.lock | 40 +++++++++++++-------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 18 ++++++------ packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 ++ packages/ai/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 ++ packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 ++ packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- 20 files changed, 59 insertions(+), 53 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 2853a76f1..30e88ac48 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2330,7 +2330,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.10.8" +version = "15.10.9" dependencies = [ "anyhow", "ast-grep-core", @@ -2398,7 +2398,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.10.8" +version = "15.10.9" dependencies = [ "async-trait", "libc", @@ -2410,7 +2410,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.10.8" +version = "15.10.9" dependencies = [ "anyhow", "arboard", @@ -2456,7 +2456,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.10.8" +version = "15.10.9" dependencies = [ "anyhow", "brush-builtins", @@ -2759,9 +2759,9 @@ dependencies = [ [[package]] name = "regex" -version = "1.12.3" +version = "1.12.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e10754a14b9137dd7b1e3e5b0493cc9171fdd105e0ab477f51b72e7f3ac0e276" +checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba" dependencies = [ "aho-corasick", "memchr", @@ -2782,9 +2782,9 @@ dependencies = [ [[package]] name = "regex-syntax" -version = "0.8.10" +version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" [[package]] name = "rgb" @@ -4129,9 +4129,9 @@ dependencies = [ [[package]] name = "uuid" -version = "1.23.2" +version = "1.23.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d258b83ceec21034727ecee8c382cfa6c3e133699b0742c64571814fb420c9f7" +checksum = "144d6b123cef80b301b8f72a9e2ca4370ddec21950d0a103dd22c437006d2db7" dependencies = [ "js-sys", "wasm-bindgen", diff --git a/Cargo.toml b/Cargo.toml index b2fce873c..fab637545 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.10.8" +version = "15.10.9" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index c86b359fc..0c35569bd 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.10.8", + "version": "15.10.9", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.10.8", + "version": "15.10.9", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -44,7 +44,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.10.8", + "version": "15.10.9", "bin": { "omp": "src/cli.ts", }, @@ -90,7 +90,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.10.8", + "version": "15.10.9", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -101,7 +101,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.10.8", + "version": "15.10.9", "bin": { "mnemopi": "src/cli.ts", }, @@ -118,7 +118,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.10.8", + "version": "15.10.9", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -126,7 +126,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.10.8", + "version": "15.10.9", "bin": { "omp-stats": "./src/index.ts", }, @@ -151,7 +151,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.10.8", + "version": "15.10.9", "bin": { "omp-swarm": "src/cli.ts", }, @@ -167,7 +167,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.10.8", + "version": "15.10.9", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -208,7 +208,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.10.8", + "version": "15.10.9", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -248,15 +248,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.10.8", - "@oh-my-pi/omp-stats": "15.10.8", - "@oh-my-pi/pi-agent-core": "15.10.8", - "@oh-my-pi/pi-ai": "15.10.8", - "@oh-my-pi/pi-coding-agent": "15.10.8", - "@oh-my-pi/pi-mnemopi": "15.10.8", - "@oh-my-pi/pi-natives": "15.10.8", - "@oh-my-pi/pi-tui": "15.10.8", - "@oh-my-pi/pi-utils": "15.10.8", + "@oh-my-pi/hashline": "15.10.9", + "@oh-my-pi/omp-stats": "15.10.9", + "@oh-my-pi/pi-agent-core": "15.10.9", + "@oh-my-pi/pi-ai": "15.10.9", + "@oh-my-pi/pi-coding-agent": "15.10.9", + "@oh-my-pi/pi-mnemopi": "15.10.9", + "@oh-my-pi/pi-natives": "15.10.9", + "@oh-my-pi/pi-tui": "15.10.9", + "@oh-my-pi/pi-utils": "15.10.9", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -849,7 +849,7 @@ "bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="], - "caniuse-lite": ["caniuse-lite@1.0.30001793", "", {}, "sha512-iwSsYWaCOoh26cV8NwNRViHlrfUvYsHDfRVcbtmw0Kg6PJIZZXwMkj1442FYLBGkeUf1juAsU3DTfxW579mrPA=="], + "caniuse-lite": ["caniuse-lite@1.0.30001797", "", {}, "sha512-l8xKG+gwAIExZGl9FrF7KUwuOmk6wbEPC9Xoy/RtnWv1XG0Q4LFlagaLpUv3Kiza3W/wm27zy0yWJEieYKAP6w=="], "chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 45408b646..8987c1979 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_10_8")] +#[napi(js_name = "__piNativesV15_10_9")] pub const fn pi_natives_version_sentinel() {} diff --git a/package.json b/package.json index d90656dc5..4489c53be 100644 --- a/package.json +++ b/package.json @@ -20,15 +20,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.10.8", - "@oh-my-pi/omp-stats": "15.10.8", - "@oh-my-pi/pi-agent-core": "15.10.8", - "@oh-my-pi/pi-ai": "15.10.8", - "@oh-my-pi/pi-coding-agent": "15.10.8", - "@oh-my-pi/pi-mnemopi": "15.10.8", - "@oh-my-pi/pi-natives": "15.10.8", - "@oh-my-pi/pi-tui": "15.10.8", - "@oh-my-pi/pi-utils": "15.10.8", + "@oh-my-pi/hashline": "15.10.9", + "@oh-my-pi/omp-stats": "15.10.9", + "@oh-my-pi/pi-agent-core": "15.10.9", + "@oh-my-pi/pi-ai": "15.10.9", + "@oh-my-pi/pi-coding-agent": "15.10.9", + "@oh-my-pi/pi-mnemopi": "15.10.9", + "@oh-my-pi/pi-natives": "15.10.9", + "@oh-my-pi/pi-tui": "15.10.9", + "@oh-my-pi/pi-utils": "15.10.9", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index 6d701e976..53bab4413 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.10.8", + "version": "15.10.9", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index aee3d68d2..c3df6ab31 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.10.9] - 2026-06-09 + ### Added - Added `antigravityRankingStrategy` and registered it as the default `CredentialRankingStrategy` for `google-antigravity`, so multi-account selection consumes the per-counter Antigravity usage reports (sorted ascending by `remainingFraction` in `fetchAntigravityUsage`) before falling back to round-robin — preventing the exhausted-counter credential from being chosen first when an unblocked sibling has headroom ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)). diff --git a/packages/ai/package.json b/packages/ai/package.json index ba1d62da4..e4a1a712e 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.10.8", + "version": "15.10.9", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index fd8a5927d..5e0433e34 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.10.9] - 2026-06-09 + ### Fixed - Fixed streaming thinking (and other styled assistant content) vanishing from native scrollback once it scrolled past the viewport top during a foreground turn. The transcript's append-only commit detector compared raw row bytes, so a styled paragraph wrapping onto a new row (the span-closing SGR and width padding move while the visible cells stay identical) or a streamed token pushing the last word down a line flagged the block as permanently volatile — the commit boundary froze and every later row that crossed the viewport top was committed nowhere. Rows are now compared by visible content, a wrap-shrink of the in-flight bottom line counts as append-only, and a genuine one-off interior rewrite only suspends commits until the block re-earns append-only (30 clean frames), after which the pinned emitter backfills the stalled gap contiguously. Periodically rewriting blocks (spinners, collapsing tool previews) never re-earn and stay deferred. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index d252cf5ed..b69b067f5 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "15.10.8", + "version": "15.10.9", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 525054ff2..e41921122 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "15.10.8", + "version": "15.10.9", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 4e08e1093..eb0339185 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "15.10.8", + "version": "15.10.9", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 6531159fd..46f1021f2 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -136,7 +136,7 @@ export declare class Shell { * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV15_10_8(): void +export declare function __piNativesV15_10_9(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 4b52ccdec..b388c6e7d 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -23,7 +23,7 @@ export const PtySession = nativeBindings.PtySession; export const Shell = nativeBindings.Shell; // functions -export const __piNativesV15_10_8 = nativeBindings.__piNativesV15_10_8; +export const __piNativesV15_10_9 = nativeBindings.__piNativesV15_10_9; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 326a404af..0ad99b0ba 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "15.10.8", + "version": "15.10.9", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/stats/package.json b/packages/stats/package.json index 61ff61456..c67998461 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "15.10.8", + "version": "15.10.9", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 7128821da..350725c28 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "15.10.8", + "version": "15.10.9", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5e864102d..36e88c3bb 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.10.9] - 2026-06-09 + ### Added - Added a `wrapDescription` option to `SelectListLayoutOptions`. When enabled, long descriptions wrap onto continuation rows indented under the description column instead of being silently truncated. The slash-command/skill autocomplete picker now opts in so descriptions like the bundled skills' remain fully readable at normal terminal widths. `maxVisible` becomes the picker's visual row budget so the popup height stays bounded even when items wrap (a single 5-row description with `maxVisible=3` clips with the scrollbar carrying the offscreen tail). Navigation stays item-to-item, the narrow-width fallback (`width <= 40`) is unchanged, and the `ScrollView` scrollbar tracks visual rows so the thumb stays correct when items wrap unevenly. ([#2169](https://github.com/can1357/oh-my-pi/issues/2169)) diff --git a/packages/tui/package.json b/packages/tui/package.json index 2bfc985f0..d5c62294e 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "15.10.8", + "version": "15.10.9", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 79b1b02a9..c4a664768 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "15.10.8", + "version": "15.10.9", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk",