From c7ec6d066d8f49a649ce54b72e34dcbb8efb201f Mon Sep 17 00:00:00 2001 From: oldschoola Date: Sun, 21 Jun 2026 18:04:42 -0700 Subject: [PATCH 01/91] feat(coding-agent): recognize # as a GitHub issue/PR reference Typing # (e.g. #3164) in the prompt now offers PR and Issue autocomplete candidates; accepting one rewrites the token to the pr:///issue:// internal URL (+ trailing space, matching the @/internal-url convention). The existing read tool -> InternalUrlRouter -> gh pipeline resolves it from the cwd's git remote, so no new resolution code is needed. Bare # and # keep the existing prompt-action menu (additive, no regression). # requires a positive integer, so #0 / leading zeros do not offer candidates. Closes #3218 --- packages/coding-agent/CHANGELOG.md | 4 + .../src/modes/github-ref-autocomplete.ts | 55 +++++++++++ .../src/modes/prompt-action-autocomplete.ts | 6 ++ .../src/modes/utils/hotkeys-markdown.ts | 3 +- .../command-controller-hotkeys.test.ts | 3 +- .../modes/github-ref-autocomplete.test.ts | 91 +++++++++++++++++++ 6 files changed, 160 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/src/modes/github-ref-autocomplete.ts create mode 100644 packages/coding-agent/test/modes/github-ref-autocomplete.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 13b3ae74d..3be8059ea 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Typing `#` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline ([#3218](https://github.com/can1357/oh-my-pi/issues/3218)) + ## [16.1.12] - 2026-06-21 ### Changed diff --git a/packages/coding-agent/src/modes/github-ref-autocomplete.ts b/packages/coding-agent/src/modes/github-ref-autocomplete.ts new file mode 100644 index 000000000..f60d9a5d5 --- /dev/null +++ b/packages/coding-agent/src/modes/github-ref-autocomplete.ts @@ -0,0 +1,55 @@ +/** + * Autocomplete for GitHub issue/PR references typed as `#` (e.g. `#3164`). + * + * Mirrors the `@` file-reference and `scheme://` internal-url conventions: the + * `#` token is rewritten to an internal URL (`pr://3164` or + * `issue://3164`) plus a trailing space, and the existing tool-mediated pipeline + * (the `read` tool → InternalUrlRouter → `gh`) resolves it from the session + * cwd's git remote. + * + * No network at suggestion time — candidates are generated locally. GitHub + * shares the issue/PR number space and there is no cheap way to tell which a + * given number is while typing, so both a PR and an Issue candidate are offered + * (PR first — the more common reference in a coding context) and the user + * disambiguates by accepting the right one. Anything that is not a pure `#` + * token keeps falling through to the existing prompt-action menu. + */ +import type { AutocompleteItem } from "@oh-my-pi/pi-tui"; + +/** Candidates offered for a `#` token, in display order. */ +const GITHUB_REF_KINDS = [ + { scheme: "pr", label: "PR", description: "GitHub pull request" }, + { scheme: "issue", label: "Issue", description: "GitHub issue" }, +] as const; + +/** + * Detect a `#` token ending at the cursor. Only a `#` followed by a + * positive integer (no leading zeros) qualifies, so `#3164` matches but `#`, + * `#0`, `#0123`, `#copy`, and `#3164abc` do not. + */ +export function getGithubRefPrefix(textBeforeCursor: string): string | null { + const hashIndex = textBeforeCursor.lastIndexOf("#"); + if (hashIndex === -1) return null; + const token = textBeforeCursor.slice(hashIndex + 1); + if (!/^[1-9]\d*$/.test(token)) return null; + return `#${token}`; +} + +/** + * Suggestions for a `#` token: a PR candidate and an Issue candidate, + * each rewriting to the corresponding internal URL on accept. Returns `null` + * when the text before the cursor is not a `#` token. + */ +export function getGithubRefSuggestions( + textBeforeCursor: string, +): { items: AutocompleteItem[]; prefix: string } | null { + const prefix = getGithubRefPrefix(textBeforeCursor); + if (!prefix) return null; + const number = prefix.slice(1); + const items: AutocompleteItem[] = GITHUB_REF_KINDS.map(kind => ({ + value: `${kind.scheme}://${number}`, + label: `${kind.label} #${number}`, + description: kind.description, + })); + return { items, prefix }; +} diff --git a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts index 1ec0dcc9c..2f9507fa8 100644 --- a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts +++ b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts @@ -8,6 +8,7 @@ import { import { formatKeyHints, type KeybindingsManager } from "../config/keybindings"; import { isSettingsInitialized, settings } from "../config/settings"; import { applyEmojiCompletion, getEmojiSuggestions, isEmojiPrefix, tryEmojiInlineReplace } from "./emoji-autocomplete"; +import { getGithubRefPrefix, getGithubRefSuggestions } from "./github-ref-autocomplete"; import { applyInternalUrlCompletion, getInternalUrlSuggestions, @@ -109,6 +110,8 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider { ): Promise<{ items: AutocompleteItem[]; prefix: string } | null> { const currentLine = lines[cursorLine] || ""; const textBeforeCursor = currentLine.slice(0, cursorCol); + const githubRefSuggestions = getGithubRefSuggestions(textBeforeCursor); + if (githubRefSuggestions) return githubRefSuggestions; const promptActionPrefix = getPromptActionPrefix(textBeforeCursor); if (promptActionPrefix) { const query = promptActionPrefix.slice(1).toLowerCase(); @@ -156,6 +159,9 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider { cursorCol: number; onApplied?: () => void; } { + if (getGithubRefPrefix(prefix)) { + return applyInternalUrlCompletion(lines, cursorLine, cursorCol, item, prefix); + } if (prefix.startsWith("#") && isPromptActionItem(item)) { if (item.actionId === "undo") { return { diff --git a/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts b/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts index f523732a1..419e91792 100644 --- a/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts +++ b/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts @@ -52,7 +52,8 @@ export function buildHotkeysMarkdown(bindings: HotkeysMarkdownBindings): string `| \`${appKey(bindings, "app.clipboard.pasteImage")}\` | Paste image or text from clipboard |`, "| Hold `Space` | Speech-to-text (push-to-talk): hold to record, release to transcribe |", `| \`${appKey(bindings, "app.agents.hub")}\` / \`${appKey(bindings, "app.session.observe")}\` / double-tap \`←\` (empty editor) | Open the agent hub |`, - "| `#` | Open prompt actions |", + "| `#` | GitHub issue/PR reference (e.g. `#3164` → `pr://`/`issue://`) |", + "| `#` / `#` | Prompt actions (copy / undo / move cursor) |", "| `/` | Slash commands |", "| `!` | Run bash command |", "| `!!` | Run bash command (excluded from context) |", diff --git a/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts b/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts index 382d2ec56..ce05a9fb5 100644 --- a/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts +++ b/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts @@ -41,7 +41,8 @@ describe("buildHotkeysMarkdown", () => { expect(markdown).toContain("| `Ctrl+L` | Reset terminal display |"); expect(markdown).toContain("| `Alt+R` | Retry last failed assistant turn |"); expect(markdown).toContain("| `Alt+Shift+P` | Toggle plan mode |"); - expect(markdown).toContain("| `#` | Open prompt actions |"); + expect(markdown).toContain("| `#` | GitHub issue/PR reference"); + expect(markdown).toContain("| `#` / `#` | Prompt actions"); for (const line of lines) { if (line.length === 0) continue; expect(line.startsWith(" ")).toBe(false); diff --git a/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts new file mode 100644 index 000000000..2d3289594 --- /dev/null +++ b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts @@ -0,0 +1,91 @@ +import { describe, expect, it } from "bun:test"; +import { KeybindingsManager as AppKeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; +import { getGithubRefPrefix, getGithubRefSuggestions } from "@oh-my-pi/pi-coding-agent/modes/github-ref-autocomplete"; +import { createPromptActionAutocompleteProvider } from "@oh-my-pi/pi-coding-agent/modes/prompt-action-autocomplete"; + +function makeProvider() { + return createPromptActionAutocompleteProvider({ + commands: [], + basePath: "/tmp", + keybindings: AppKeybindingsManager.inMemory({}), + copyCurrentLine: () => {}, + copyPrompt: () => {}, + undo: () => {}, + moveCursorToMessageEnd: () => {}, + moveCursorToMessageStart: () => {}, + moveCursorToLineStart: () => {}, + moveCursorToLineEnd: () => {}, + }); +} + +describe("github-ref autocomplete — prefix detection", () => { + it("matches the last # token ending at the cursor", () => { + expect(getGithubRefPrefix("#3164")).toBe("#3164"); + expect(getGithubRefPrefix("look at #3164")).toBe("#3164"); + expect(getGithubRefPrefix("see #1 and #3164")).toBe("#3164"); + }); + + it("does not match bare #, text, or mixed tokens", () => { + expect(getGithubRefPrefix("#")).toBeNull(); + expect(getGithubRefPrefix("#copy")).toBeNull(); + expect(getGithubRefPrefix("#3164abc")).toBeNull(); + expect(getGithubRefPrefix("#3a")).toBeNull(); + // zero / leading zeros are not valid GitHub numbers + expect(getGithubRefPrefix("#0")).toBeNull(); + expect(getGithubRefPrefix("#00")).toBeNull(); + expect(getGithubRefPrefix("#0123")).toBeNull(); + // a space after the digits closes the token + expect(getGithubRefPrefix("#3164 ")).toBeNull(); + expect(getGithubRefPrefix("no hash here")).toBeNull(); + }); +}); + +describe("github-ref autocomplete — suggestions", () => { + it("offers a PR and an Issue candidate for #", () => { + const result = getGithubRefSuggestions("#3164"); + expect(result).not.toBeNull(); + expect(result!.prefix).toBe("#3164"); + expect(result!.items).toEqual([ + { value: "pr://3164", label: "PR #3164", description: "GitHub pull request" }, + { value: "issue://3164", label: "Issue #3164", description: "GitHub issue" }, + ]); + }); + + it("returns null for non-numeric tokens", () => { + expect(getGithubRefSuggestions("#copy")).toBeNull(); + expect(getGithubRefSuggestions("#")).toBeNull(); + expect(getGithubRefSuggestions("#3164abc")).toBeNull(); + expect(getGithubRefSuggestions("#0")).toBeNull(); + }); +}); + +describe("github-ref autocomplete — provider integration", () => { + it("yields the ref candidates and rewrites the token to the chosen internal URL", async () => { + const provider = makeProvider(); + const suggestions = await provider.getSuggestions(["review #3164"], 0, 12); + expect(suggestions).not.toBeNull(); + expect(suggestions!.prefix).toBe("#3164"); + expect(suggestions!.items.map(item => item.value)).toEqual(["pr://3164", "issue://3164"]); + + const pr = suggestions!.items[0]!; + const issue = suggestions!.items[1]!; + + const prResult = provider.applyCompletion(["review #3164"], 0, 12, pr, suggestions!.prefix); + expect(prResult.lines).toEqual(["review pr://3164 "]); + expect(prResult.cursorCol).toBe("review pr://3164 ".length); + + const issueResult = provider.applyCompletion(["review #3164"], 0, 12, issue, suggestions!.prefix); + expect(issueResult.lines).toEqual(["review issue://3164 "]); + }); + + it("leaves # and bare # to the prompt-action menu (no github-ref candidates)", async () => { + const provider = makeProvider(); + const isRef = (value: string) => value.startsWith("pr://") || value.startsWith("issue://"); + + const textSuggestions = await provider.getSuggestions(["#copy"], 0, 5); + expect(textSuggestions?.items.every(item => !isRef(item.value))).toBe(true); + + const bareSuggestions = await provider.getSuggestions(["#"], 0, 1); + expect(bareSuggestions?.items.every(item => !isRef(item.value))).toBe(true); + }); +}); From d8f52e8a9b1ce931f40645c0094d6d0029a6c736 Mon Sep 17 00:00:00 2001 From: oldschoola Date: Sun, 21 Jun 2026 18:36:20 -0700 Subject: [PATCH 02/91] fix(coding-agent): require token boundary for #; add pr/issue qualifier Address review feedback on #3224 (roboomp + Codex): # now requires a token boundary before # (start/whitespace/quote/paren/` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline ([#3218](https://github.com/can1357/oh-my-pi/issues/3218)) +- Typing `#` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline. Naming the type (`pr #3164` / `issue #3164`) constrains the candidates to that kind, and embedded hashes like `owner/repo#N`, `foo#N`, or URL fragments are left untouched ([#3218](https://github.com/can1357/oh-my-pi/issues/3218)) ## [16.1.12] - 2026-06-21 diff --git a/packages/coding-agent/src/modes/github-ref-autocomplete.ts b/packages/coding-agent/src/modes/github-ref-autocomplete.ts index f60d9a5d5..0b261f480 100644 --- a/packages/coding-agent/src/modes/github-ref-autocomplete.ts +++ b/packages/coding-agent/src/modes/github-ref-autocomplete.ts @@ -2,54 +2,74 @@ * Autocomplete for GitHub issue/PR references typed as `#` (e.g. `#3164`). * * Mirrors the `@` file-reference and `scheme://` internal-url conventions: the - * `#` token is rewritten to an internal URL (`pr://3164` or - * `issue://3164`) plus a trailing space, and the existing tool-mediated pipeline - * (the `read` tool → InternalUrlRouter → `gh`) resolves it from the session - * cwd's git remote. + * token is rewritten to an internal URL (`pr://3164` or `issue://3164`) plus a + * trailing space, and the existing tool-mediated pipeline (the `read` tool → + * InternalUrlRouter → `gh`) resolves it from the session cwd's git remote. * * No network at suggestion time — candidates are generated locally. GitHub * shares the issue/PR number space and there is no cheap way to tell which a * given number is while typing, so both a PR and an Issue candidate are offered - * (PR first — the more common reference in a coding context) and the user - * disambiguates by accepting the right one. Anything that is not a pure `#` - * token keeps falling through to the existing prompt-action menu. + * by default. Naming the type first (`pr #3164` / `issue #3164`) constrains the + * candidates to that kind. Anything that is not a standalone `#` token + * keeps falling through to the existing prompt-action menu. */ import type { AutocompleteItem } from "@oh-my-pi/pi-tui"; -/** Candidates offered for a `#` token, in display order. */ +/** Candidate kinds, in default display order. */ const GITHUB_REF_KINDS = [ - { scheme: "pr", label: "PR", description: "GitHub pull request" }, - { scheme: "issue", label: "Issue", description: "GitHub issue" }, + { qualifier: "pr", scheme: "pr", label: "PR", description: "GitHub pull request" }, + { qualifier: "issue", scheme: "issue", label: "Issue", description: "GitHub issue" }, ] as const; -/** - * Detect a `#` token ending at the cursor. Only a `#` followed by a - * positive integer (no leading zeros) qualifies, so `#3164` matches but `#`, - * `#0`, `#0123`, `#copy`, and `#3164abc` do not. - */ -export function getGithubRefPrefix(textBeforeCursor: string): string | null { - const hashIndex = textBeforeCursor.lastIndexOf("#"); - if (hashIndex === -1) return null; - const token = textBeforeCursor.slice(hashIndex + 1); - if (!/^[1-9]\d*$/.test(token)) return null; - return `#${token}`; +export interface GithubRefContext { + /** Text to replace on accept: `#3164`, or `pr #3164` when a qualifier precedes it. */ + prefix: string; + /** Type the user named (`pr`/`pull` → `pr`, `issue` → `issue`), or null to offer both. */ + qualifier: "pr" | "issue" | null; + /** The numeric reference, e.g. `3164`. */ + number: string; } /** - * Suggestions for a `#` token: a PR candidate and an Issue candidate, - * each rewriting to the corresponding internal URL on accept. Returns `null` - * when the text before the cursor is not a `#` token. + * A standalone `#` token ending at the cursor. The `#` must be + * preceded by a token boundary (start, whitespace, or an opening quote/paren/`<`/`=`, + * matching the internal-URL boundary set) so embedded hashes like `owner/repo#N`, + * `foo#N`, `C#12`, or a URL fragment do not match. An optional `pr`/`pull`/`issue` + * qualifier word (case-insensitive) immediately before the `#` constrains the kind. + */ +const GITHUB_REF_TOKEN_RE = /(?:^|[\s"'`(<=])(?:(pr|pull|issue)(\s+))?#([1-9]\d*)$/i; + +export function getGithubRefContext(textBeforeCursor: string): GithubRefContext | null { + const match = textBeforeCursor.match(GITHUB_REF_TOKEN_RE); + if (!match) return null; + const qualifierWord = match[1]; + const whitespace = match[2] ?? ""; + const number = match[3] ?? ""; + return { + prefix: qualifierWord ? `${qualifierWord}${whitespace}#${number}` : `#${number}`, + qualifier: !qualifierWord ? null : qualifierWord.toLowerCase() === "issue" ? "issue" : "pr", + number, + }; +} + +/** + * Suggestions for a `#` token. Both kinds are offered unless the user + * named a type (`pr #3164` / `issue #3164`), in which case only that kind is + * offered. Returns `null` when the text before the cursor is not a standalone + * `#` token. */ export function getGithubRefSuggestions( textBeforeCursor: string, ): { items: AutocompleteItem[]; prefix: string } | null { - const prefix = getGithubRefPrefix(textBeforeCursor); - if (!prefix) return null; - const number = prefix.slice(1); - const items: AutocompleteItem[] = GITHUB_REF_KINDS.map(kind => ({ - value: `${kind.scheme}://${number}`, - label: `${kind.label} #${number}`, + const context = getGithubRefContext(textBeforeCursor); + if (!context) return null; + const kinds = context.qualifier + ? GITHUB_REF_KINDS.filter(kind => kind.qualifier === context.qualifier) + : GITHUB_REF_KINDS; + const items: AutocompleteItem[] = kinds.map(kind => ({ + value: `${kind.scheme}://${context.number}`, + label: `${kind.label} #${context.number}`, description: kind.description, })); - return { items, prefix }; + return { items, prefix: context.prefix }; } diff --git a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts index 2f9507fa8..eb03c1b01 100644 --- a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts +++ b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts @@ -8,7 +8,7 @@ import { import { formatKeyHints, type KeybindingsManager } from "../config/keybindings"; import { isSettingsInitialized, settings } from "../config/settings"; import { applyEmojiCompletion, getEmojiSuggestions, isEmojiPrefix, tryEmojiInlineReplace } from "./emoji-autocomplete"; -import { getGithubRefPrefix, getGithubRefSuggestions } from "./github-ref-autocomplete"; +import { getGithubRefContext, getGithubRefSuggestions } from "./github-ref-autocomplete"; import { applyInternalUrlCompletion, getInternalUrlSuggestions, @@ -159,7 +159,7 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider { cursorCol: number; onApplied?: () => void; } { - if (getGithubRefPrefix(prefix)) { + if (getGithubRefContext(prefix)) { return applyInternalUrlCompletion(lines, cursorLine, cursorCol, item, prefix); } if (prefix.startsWith("#") && isPromptActionItem(item)) { diff --git a/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts index 2d3289594..ba77380fa 100644 --- a/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts +++ b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from "bun:test"; import { KeybindingsManager as AppKeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; -import { getGithubRefPrefix, getGithubRefSuggestions } from "@oh-my-pi/pi-coding-agent/modes/github-ref-autocomplete"; +import { getGithubRefContext, getGithubRefSuggestions } from "@oh-my-pi/pi-coding-agent/modes/github-ref-autocomplete"; import { createPromptActionAutocompleteProvider } from "@oh-my-pi/pi-coding-agent/modes/prompt-action-autocomplete"; function makeProvider() { @@ -18,30 +18,62 @@ function makeProvider() { }); } -describe("github-ref autocomplete — prefix detection", () => { - it("matches the last # token ending at the cursor", () => { - expect(getGithubRefPrefix("#3164")).toBe("#3164"); - expect(getGithubRefPrefix("look at #3164")).toBe("#3164"); - expect(getGithubRefPrefix("see #1 and #3164")).toBe("#3164"); +describe("github-ref autocomplete — token detection", () => { + it("matches a standalone # ending at the cursor", () => { + expect(getGithubRefContext("#3164")).toEqual({ prefix: "#3164", qualifier: null, number: "3164" }); + expect(getGithubRefContext("look at #3164")).toEqual({ prefix: "#3164", qualifier: null, number: "3164" }); + // only the token ending at the cursor (the $ anchor) wins + expect(getGithubRefContext("see #1 then #3164")).toEqual({ + prefix: "#3164", + qualifier: null, + number: "3164", + }); }); - it("does not match bare #, text, or mixed tokens", () => { - expect(getGithubRefPrefix("#")).toBeNull(); - expect(getGithubRefPrefix("#copy")).toBeNull(); - expect(getGithubRefPrefix("#3164abc")).toBeNull(); - expect(getGithubRefPrefix("#3a")).toBeNull(); - // zero / leading zeros are not valid GitHub numbers - expect(getGithubRefPrefix("#0")).toBeNull(); - expect(getGithubRefPrefix("#00")).toBeNull(); - expect(getGithubRefPrefix("#0123")).toBeNull(); + it("requires a token boundary before # so embedded hashes don't match", () => { + // cross-repo reference, mid-word, URL fragment — none should offer candidates + expect(getGithubRefContext("owner/repo#3164")).toBeNull(); + expect(getGithubRefContext("foo#3164")).toBeNull(); + expect(getGithubRefContext("C#12")).toBeNull(); + expect(getGithubRefContext("https://github.com/can1357/oh-my-pi#3164")).toBeNull(); + expect(getGithubRefContext("path/#3164")).toBeNull(); + }); + + it("does not match bare #, text, mixed, zero, or leading zeros", () => { + expect(getGithubRefContext("#")).toBeNull(); + expect(getGithubRefContext("#copy")).toBeNull(); + expect(getGithubRefContext("#3164abc")).toBeNull(); + expect(getGithubRefContext("#3a")).toBeNull(); // a space after the digits closes the token - expect(getGithubRefPrefix("#3164 ")).toBeNull(); - expect(getGithubRefPrefix("no hash here")).toBeNull(); + expect(getGithubRefContext("#3164 ")).toBeNull(); + expect(getGithubRefContext("#0")).toBeNull(); + expect(getGithubRefContext("#0123")).toBeNull(); + }); + + it("detects a pr/pull/issue qualifier word immediately before the number", () => { + expect(getGithubRefContext("pr #3164")).toEqual({ prefix: "pr #3164", qualifier: "pr", number: "3164" }); + expect(getGithubRefContext("PR #3164")).toEqual({ prefix: "PR #3164", qualifier: "pr", number: "3164" }); + expect(getGithubRefContext("pull #3164")).toEqual({ prefix: "pull #3164", qualifier: "pr", number: "3164" }); + expect(getGithubRefContext("issue #3164")).toEqual({ + prefix: "issue #3164", + qualifier: "issue", + number: "3164", + }); + // the word right before the number is the qualifier, even with leading text + expect(getGithubRefContext("look at the issue #3164")?.qualifier).toBe("issue"); + }); + + it("does not treat arbitrary words, or qualifiers glued to a path, as the type", () => { + expect(getGithubRefContext("review #3164")?.qualifier).toBeNull(); + // "pr" inside "src/pr" is preceded by '/', not a boundary, so it is not a qualifier + expect(getGithubRefContext("src/pr #3164")?.qualifier).toBeNull(); + // a qualifier with no space before the # is not recognized + expect(getGithubRefContext("pr#3164")).toBeNull(); }); }); describe("github-ref autocomplete — suggestions", () => { - it("offers a PR and an Issue candidate for #", () => { + it("offers both candidates when no qualifier is given", () => { const result = getGithubRefSuggestions("#3164"); expect(result).not.toBeNull(); expect(result!.prefix).toBe("#3164"); @@ -51,16 +83,26 @@ describe("github-ref autocomplete — suggestions", () => { ]); }); - it("returns null for non-numeric tokens", () => { + it("offers only the PR candidate for a pr/pull qualifier", () => { + const result = getGithubRefSuggestions("pr #3164"); + expect(result!.prefix).toBe("pr #3164"); + expect(result!.items).toEqual([{ value: "pr://3164", label: "PR #3164", description: "GitHub pull request" }]); + }); + + it("offers only the Issue candidate for an issue qualifier", () => { + const result = getGithubRefSuggestions("issue #3164"); + expect(result!.items).toEqual([{ value: "issue://3164", label: "Issue #3164", description: "GitHub issue" }]); + }); + + it("returns null for embedded or non-ref text", () => { + expect(getGithubRefSuggestions("owner/repo#3164")).toBeNull(); expect(getGithubRefSuggestions("#copy")).toBeNull(); - expect(getGithubRefSuggestions("#")).toBeNull(); - expect(getGithubRefSuggestions("#3164abc")).toBeNull(); expect(getGithubRefSuggestions("#0")).toBeNull(); }); }); describe("github-ref autocomplete — provider integration", () => { - it("yields the ref candidates and rewrites the token to the chosen internal URL", async () => { + it("yields both candidates and rewrites the token to the chosen internal URL", async () => { const provider = makeProvider(); const suggestions = await provider.getSuggestions(["review #3164"], 0, 12); expect(suggestions).not.toBeNull(); @@ -69,7 +111,6 @@ describe("github-ref autocomplete — provider integration", () => { const pr = suggestions!.items[0]!; const issue = suggestions!.items[1]!; - const prResult = provider.applyCompletion(["review #3164"], 0, 12, pr, suggestions!.prefix); expect(prResult.lines).toEqual(["review pr://3164 "]); expect(prResult.cursorCol).toBe("review pr://3164 ".length); @@ -78,14 +119,23 @@ describe("github-ref autocomplete — provider integration", () => { expect(issueResult.lines).toEqual(["review issue://3164 "]); }); - it("leaves # and bare # to the prompt-action menu (no github-ref candidates)", async () => { + it("constrains to the named type and consumes the qualifier on accept", async () => { + const provider = makeProvider(); + const suggestions = await provider.getSuggestions(["review pr #3164"], 0, 15); + expect(suggestions).not.toBeNull(); + expect(suggestions!.prefix).toBe("pr #3164"); + expect(suggestions!.items.map(item => item.value)).toEqual(["pr://3164"]); + + const pr = suggestions!.items[0]!; + const result = provider.applyCompletion(["review pr #3164"], 0, 15, pr, suggestions!.prefix); + // the "pr " qualifier is replaced along with the number, not left dangling + expect(result.lines).toEqual(["review pr://3164 "]); + }); + + it("does not offer candidates for embedded hashes (falls through to other providers)", async () => { const provider = makeProvider(); const isRef = (value: string) => value.startsWith("pr://") || value.startsWith("issue://"); - - const textSuggestions = await provider.getSuggestions(["#copy"], 0, 5); - expect(textSuggestions?.items.every(item => !isRef(item.value))).toBe(true); - - const bareSuggestions = await provider.getSuggestions(["#"], 0, 1); - expect(bareSuggestions?.items.every(item => !isRef(item.value))).toBe(true); + const embedded = await provider.getSuggestions(["owner/repo#3164"], 0, 15); + expect(embedded?.items.every(item => !isRef(item.value)) ?? true).toBe(true); }); }); From 1ac9802508060e136c28575255f3824044e383cd Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 4 Jul 2026 16:42:38 +0000 Subject: [PATCH 03/91] fix(coding-agent): restored fallback model selection - Exposed retry fallback chains in the model settings panel. - Added a /model action that assigns the selected model as the default retry fallback. - Cleared retry cooldown suppression when users manually switch models. Fixes #4533 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/config/model-registry.ts | 9 +++ .../src/config/settings-schema.ts | 12 +++- .../src/modes/components/model-selector.ts | 36 ++++++++-- .../src/modes/components/settings-defs.ts | 2 +- .../modes/controllers/selector-controller.ts | 16 ++++- .../coding-agent/src/session/agent-session.ts | 4 ++ ...model-selector-role-badge-thinking.test.ts | 65 ++++++++++++++++++- .../modes/components/settings-layout.test.ts | 18 +++++ 9 files changed, 155 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9c2395e3d..40eaff9d9 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533)) + ## [16.3.6] - 2026-07-04 ### Changed diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 0b5854dbc..6b1a3da76 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -2214,6 +2214,15 @@ export class ModelRegistry { return true; } + /** + * Clear the cooldown suppression for one selector after an explicit user selection. + */ + clearSuppressedSelector(selector: string): void { + this.#suppressedSelectors.delete( + normalizeSuppressedSelector(selector, (provider, id) => this.find(provider, id) !== undefined), + ); + } + /** * Clear all cooldown suppressions recorded via {@link suppressSelector}. * Used to reset retry-fallback cooldown state without a full {@link refresh}. diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 2316034da..f8b44ba57 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -1367,7 +1367,17 @@ export const SETTINGS_SCHEMA = { description: "Allow retry recovery to switch to configured fallback models", }, }, - "retry.fallbackChains": { type: "record", default: {} as Record }, + "retry.fallbackChains": { + type: "record", + default: {} as Record, + ui: { + tab: "model", + group: "Retry & Fallback", + label: "Retry Fallback Chains", + description: + 'JSON object mapping model roles to ordered fallback model selectors, e.g. {"default":["openai/gpt-4o-mini"]}.', + }, + }, "retry.fallbackRevertPolicy": { type: "enum", values: ["cooldown-expiry", "never"] as const, diff --git a/packages/coding-agent/src/modes/components/model-selector.ts b/packages/coding-agent/src/modes/components/model-selector.ts index 8e34aa5cf..40a9e9814 100644 --- a/packages/coding-agent/src/modes/components/model-selector.ts +++ b/packages/coding-agent/src/modes/components/model-selector.ts @@ -90,16 +90,20 @@ interface RoleAssignment { autoSelected: boolean; } +type ModelSelectorAction = "modelRole" | "retryFallback"; + type RoleSelectCallback = ( model: Model, role: string | null, thinkingLevel?: ConfiguredThinkingLevel, selector?: string, + action?: ModelSelectorAction, ) => void; type CancelCallback = () => void; interface MenuRoleAction { label: string; - role: string; // now accepts custom role strings + role: string; + action: ModelSelectorAction; } interface ProviderTabState { @@ -284,14 +288,19 @@ export class ModelSelectorComponent extends Container { } #buildMenuRoleActions(): void { - this.#menuRoleActions = getKnownRoleIds(this.#settings).map(role => { + const roleActions = getKnownRoleIds(this.#settings).map(role => { const roleInfo = getRoleInfo(role, this.#settings); const roleLabel = roleInfo.tag ? `${roleInfo.tag} (${roleInfo.name})` : roleInfo.name; return { label: `Set as ${roleLabel}`, role, + action: "modelRole" as const, }; }); + this.#menuRoleActions = [ + ...roleActions, + { label: "Set as DEFAULT retry fallback", role: "default", action: "retryFallback" }, + ]; } #loadRoleModels(autoCandidateModels?: ReadonlyArray): void { @@ -1195,6 +1204,11 @@ export class ModelSelectorComponent extends Container { if (this.#menuStep === "role") { const action = this.#menuRoleActions[this.#menuSelectedIndex]; if (!action) return; + if (action.action === "retryFallback") { + this.#handleSelect(selectedItem, action.role, undefined, action.action); + this.#closeMenu(); + return; + } this.#menuSelectedRole = action.role; this.#menuStep = "thinking"; this.#menuSelectedIndex = this.#getThinkingPreselectIndex(action.role, selectedItem.model); @@ -1206,7 +1220,7 @@ export class ModelSelectorComponent extends Container { const thinkingOptions = this.#getThinkingLevelsForModel(selectedItem.model); const thinkingLevel = thinkingOptions[this.#menuSelectedIndex]; if (!thinkingLevel) return; - this.#handleSelect(selectedItem, this.#menuSelectedRole, thinkingLevel); + this.#handleSelect(selectedItem, this.#menuSelectedRole, thinkingLevel, "modelRole"); this.#closeMenu(); return; } @@ -1225,13 +1239,23 @@ export class ModelSelectorComponent extends Container { } } - #handleSelect(item: ModelItem, role: string | null, thinkingLevel?: ConfiguredThinkingLevel): void { + #handleSelect( + item: ModelItem, + role: string | null, + thinkingLevel?: ConfiguredThinkingLevel, + action: ModelSelectorAction = "modelRole", + ): void { if (this.#isItemDisabled(item)) { return; } // For temporary role, don't save to settings - just notify caller if (role === null) { - this.#onSelectCallback(item.model, null, undefined, item.selector); + this.#onSelectCallback(item.model, null, undefined, item.selector, action); + return; + } + + if (action === "retryFallback") { + this.#onSelectCallback(item.model, role, undefined, item.selector, action); return; } @@ -1241,7 +1265,7 @@ export class ModelSelectorComponent extends Container { this.#roles[role] = { model: item.model, thinkingLevel: selectedThinkingLevel, autoSelected: false }; // Notify caller (for updating agent state if needed) - this.#onSelectCallback(item.model, role, selectedThinkingLevel, item.selector); + this.#onSelectCallback(item.model, role, selectedThinkingLevel, item.selector, action); // Update list to show new badges this.#updateList(); diff --git a/packages/coding-agent/src/modes/components/settings-defs.ts b/packages/coding-agent/src/modes/components/settings-defs.ts index 582877281..dcad6310d 100644 --- a/packages/coding-agent/src/modes/components/settings-defs.ts +++ b/packages/coding-agent/src/modes/components/settings-defs.ts @@ -180,7 +180,7 @@ function pathToSettingDef(path: SettingPath): SettingDef | null { } if (schemaType === "record") { - return path === "providers.maxInFlightRequests" ? { ...base, type: "providerLimits" } : null; + return path === "providers.maxInFlightRequests" ? { ...base, type: "providerLimits" } : { ...base, type: "text" }; } return null; diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index b52d72668..2deeeb1fa 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -593,13 +593,27 @@ export class SelectorController { this.ctx.settings, this.ctx.session.modelRegistry, this.ctx.session.scopedModels, - async (model, role, thinkingLevel, selector) => { + async (model, role, thinkingLevel, selector, action) => { // `auto` is session-global: never baked into a per-role model value // (it can't round-trip through `model:`). Apply it to the session // separately and persist via `defaultThinkingLevel`. const isAuto = thinkingLevel === AUTO_THINKING; const concreteThinking = isAuto ? undefined : thinkingLevel; + const selectorValue = selector ?? `${model.provider}/${model.id}`; try { + if (action === "retryFallback" && role !== null) { + const fallbackSelector = formatModelSelectorValue(selectorValue, concreteThinking); + const fallbackChains = this.ctx.settings.get("retry.fallbackChains"); + const chain = fallbackChains[role] ?? []; + this.ctx.settings.set("retry.fallbackChains", { + ...fallbackChains, + [role]: [fallbackSelector, ...chain.filter(existing => existing !== fallbackSelector)], + }); + const roleInfo = getRoleInfo(role, settings); + const roleLabel = roleInfo?.name ?? role; + this.ctx.showStatus(`${roleLabel} fallback model: ${fallbackSelector}`); + return; + } if (role === null) { // Temporary: update agent state but don't persist the model to settings await this.ctx.session.setModelTemporary(model); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 38edd6812..54c5b9481 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -8678,6 +8678,7 @@ export class AgentSession { const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model); + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(targetModel); this.sessionManager.appendModelChange(`${targetModel.provider}/${targetModel.id}`, role); @@ -8715,6 +8716,7 @@ export class AgentSession { const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model); + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(targetModel); this.sessionManager.appendModelChange( @@ -8874,6 +8876,7 @@ export class AgentSession { const next = scopedModels[nextIndex]; // Apply model + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(next.model)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(next.model); this.sessionManager.appendModelChange(`${next.model.provider}/${next.model.id}`); @@ -8904,6 +8907,7 @@ export class AgentSession { throw new Error(`No API key for ${nextModel.provider}/${nextModel.id}`); } + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(nextModel)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(nextModel); this.sessionManager.appendModelChange(`${nextModel.provider}/${nextModel.id}`); diff --git a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts index 3e74dc631..50db05d61 100644 --- a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts +++ b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts @@ -14,6 +14,35 @@ function normalizeRenderedText(text: string): string { return stripVTControlCharacters(text).replace(/\s+/g, " ").trim(); } +const DEFAULT_RETRY_FALLBACK_ACTION_LABEL = "Set as DEFAULT retry fallback"; +const DEFAULT_RETRY_FALLBACK_ACTION = "retryFallback"; + +type ModelSelectorAction = "modelRole" | typeof DEFAULT_RETRY_FALLBACK_ACTION; +type TestRoleSelectArgs = [ + model: Model, + role: string | null, + thinkingLevel?: ConfiguredThinkingLevel, + selector?: string, + action?: ModelSelectorAction, +]; +type TestRoleSelectCallback = (...args: TestRoleSelectArgs) => void; + +function isSelectedMenuLine(line: string): boolean { + const trimmed = line.trimStart(); + return trimmed.startsWith("❯") || trimmed.startsWith("▸") || trimmed.startsWith(">") || trimmed.startsWith("\uf054"); +} + +function selectMenuAction(selector: ModelSelectorComponent, label: string): void { + for (let attempt = 0; attempt < 20; attempt++) { + const selectedTarget = stripVTControlCharacters(selector.render(220).join("\n")) + .split("\n") + .find(line => line.includes(label) && isSelectedMenuLine(line)); + if (selectedTarget) return; + selector.handleInput("\x1b[B"); + } + throw new Error(`Menu action not selectable: ${label}`); +} + function createSelector(model: Model, settings: Settings): ModelSelectorComponent { const modelRegistry = { getAll: () => [model], @@ -66,7 +95,7 @@ function createContextTestModel(id: string, contextWindow: number): Model { function createScopedSelector( models: Model[], settings: Settings, - onSelect: (model: Model, role: string | null, thinkingLevel?: ConfiguredThinkingLevel, selector?: string) => void, + onSelect: TestRoleSelectCallback, options?: { temporaryOnly?: boolean; currentContextTokens?: number }, ): ModelSelectorComponent { const modelRegistry = { @@ -82,7 +111,13 @@ function createScopedSelector( settings, modelRegistry, models.map(model => ({ model })), - (model, role, thinkingLevel, selector) => onSelect(model, role, thinkingLevel, selector), + ( + model: Model, + role: string | null, + thinkingLevel?: ConfiguredThinkingLevel, + selector?: string, + action?: ModelSelectorAction, + ) => onSelect(model, role, thinkingLevel, selector, action), () => {}, options, ); @@ -299,6 +334,32 @@ describe("ModelSelector role badge thinking display", () => { expect(onSelect.mock.calls[0]?.[3]).toBe("test/only-small"); }); + test("assigns selected model as default retry fallback without opening thinking options", () => { + installTestTheme(); + const settings = Settings.isolated({}); + const fallback = createContextTestModel("retry-fallback-model", 128_000); + const onSelect = vi.fn(); + const selector = createScopedSelector([fallback], settings, onSelect); + installTestTheme(); + + selector.handleInput("\n"); + const menuRendered = normalizeRenderedText(selector.render(220).join("\n")); + expect(menuRendered).toContain("Action for: retry-fallback-model"); + expect(menuRendered).toContain(DEFAULT_RETRY_FALLBACK_ACTION_LABEL); + + selectMenuAction(selector, DEFAULT_RETRY_FALLBACK_ACTION_LABEL); + selector.handleInput("\n"); + + const afterEnter = normalizeRenderedText(selector.render(220).join("\n")); + expect(afterEnter).not.toContain("Thinking for:"); + expect(onSelect).toHaveBeenCalledTimes(1); + const call = onSelect.mock.calls[0]; + expect(call?.[0]).toBe(fallback); + expect(call?.[1]).toBe("default"); + expect(call?.[3]).toBe("test/retry-fallback-model"); + expect(call?.[4]).toBe(DEFAULT_RETRY_FALLBACK_ACTION); + }); + test("uses cached models for Enter while offline refresh is still pending", () => { installTestTheme(); const settings = Settings.isolated({}); diff --git a/packages/coding-agent/test/modes/components/settings-layout.test.ts b/packages/coding-agent/test/modes/components/settings-layout.test.ts index dbcbd1b0e..7b657f1b5 100644 --- a/packages/coding-agent/test/modes/components/settings-layout.test.ts +++ b/packages/coding-agent/test/modes/components/settings-layout.test.ts @@ -97,4 +97,22 @@ describe("settings layout", () => { group: "Services", }); }); + + it("exposes retry fallback chains as editable JSON in the model settings", () => { + const def = getSettingsForTab("model").find(item => item.path === "retry.fallbackChains"); + + expect(def).toMatchObject({ + path: "retry.fallbackChains", + type: "text", + tab: "model", + group: "Retry & Fallback", + label: "Retry Fallback Chains", + }); + if (!def) throw new Error("retry.fallbackChains setting definition missing"); + + const description = def.description.toLowerCase(); + expect(description).toContain("json"); + expect(description).toContain("fallback"); + expect(description).toContain("selector"); + }); }); From ed7d1cb982780be4bc0ddc5a28752010e9ae8382 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 4 Jul 2026 16:54:40 +0000 Subject: [PATCH 04/91] fix(coding-agent): honored implicit fallback primary - Treated the active model as the default retry primary when only retry.fallbackChains.default is configured. - Covered the first-run case where modelRoles.default is unset but a default fallback chain exists. Fixes #4533 --- .../coding-agent/src/session/agent-session.ts | 42 +++++++++++++-- .../test/agent-session-retry-fallback.test.ts | 51 +++++++++++++++++++ 2 files changed, 89 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 54c5b9481..e44bd7d81 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -13096,6 +13096,7 @@ export class AgentSession { #resolveRetryFallbackRole(currentSelector: string): string | undefined { const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); if (!parsedCurrent) return undefined; + const chains = this.#getRetryFallbackChains(); const currentBaseSelector = formatRetryFallbackBaseSelector(parsedCurrent); const currentPlainSelector = this.model ? formatModelSelectorValue(formatModelString(this.model), parsedCurrent.thinkingLevel) @@ -13105,11 +13106,11 @@ export class AgentSession { ? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent) : undefined; - for (const role of Object.keys(this.#getRetryFallbackChains())) { + for (const role of Object.keys(chains)) { const primarySelector = this.#getRetryFallbackPrimarySelector(role); if (primarySelector?.raw === currentSelector) return role; } - for (const role of Object.keys(this.#getRetryFallbackChains())) { + for (const role of Object.keys(chains)) { const primarySelector = this.#getRetryFallbackPrimarySelector(role); if (!primarySelector) continue; if (currentPlainSelector && primarySelector.raw === currentPlainSelector) return role; @@ -13117,6 +13118,17 @@ export class AgentSession { if (primaryBaseSelector === currentBaseSelector) return role; if (currentPlainBaseSelector && primaryBaseSelector === currentPlainBaseSelector) return role; } + const defaultChain = chains.default; + if ( + Array.isArray(defaultChain) && + defaultChain.length > 0 && + this.#getRetryFallbackPrimarySelector("default") === undefined && + Object.entries(chains).every( + ([role, chain]) => role === "default" || !Array.isArray(chain) || chain.length === 0, + ) + ) { + return "default"; + } return undefined; } @@ -13135,9 +13147,31 @@ export class AgentSession { } #findRetryFallbackCandidates(role: string, currentSelector: string): RetryFallbackSelector[] { - const chain = this.#getRetryFallbackEffectiveChain(role); - if (chain.length <= 1) return []; + let chain = this.#getRetryFallbackEffectiveChain(role); const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); + if (chain.length === 0 && role === "default" && parsedCurrent) { + const chains = this.#getRetryFallbackChains(); + const defaultChain = chains.default; + if ( + Array.isArray(defaultChain) && + defaultChain.length > 0 && + this.#getRetryFallbackPrimarySelector("default") === undefined && + Object.entries(chains).every( + ([chainRole, roleChain]) => + chainRole === "default" || !Array.isArray(roleChain) || roleChain.length === 0, + ) + ) { + const seen = new Set([parsedCurrent.raw]); + chain = [parsedCurrent]; + for (const selector of defaultChain) { + const parsed = parseRetryFallbackSelector(selector, this.#modelRegistry); + if (!parsed || seen.has(parsed.raw)) continue; + seen.add(parsed.raw); + chain.push(parsed); + } + } + } + if (chain.length <= 1) return []; const currentBaseSelector = parsedCurrent ? formatRetryFallbackBaseSelector(parsedCurrent) : undefined; const currentPlainSelector = this.model && parsedCurrent diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 4ae3b5c1d..1adeda0ec 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -219,6 +219,57 @@ describe("AgentSession retry fallback", () => { ]); }); + it("uses the active initial model as the default fallback primary when the default role is unset", async () => { + const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + const fallbackModel = getBundledModel("openai", "gpt-4o-mini"); + if (!primaryModel || !fallbackModel) { + throw new Error("Expected bundled test models to exist"); + } + + const requestedModels: string[] = []; + const fallbackAppliedEvents: Array> = []; + const agent = createFallbackAgent(primaryModel, requestedModels); + + const settings = Settings.isolated({ + "compaction.enabled": false, + "retry.maxRetries": 1, + "retry.fallbackChains": { + default: [`${fallbackModel.provider}/${fallbackModel.id}`], + }, + }); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + + session.subscribe(event => { + if (event.type === "retry_fallback_applied") { + fallbackAppliedEvents.push(event); + } + }); + + await session.prompt("Recover using implicit default primary"); + await session.waitForIdle(); + + expect(requestedModels).toEqual([ + `${primaryModel.provider}/${primaryModel.id}`, + `${fallbackModel.provider}/${fallbackModel.id}`, + ]); + expect(session.model?.provider).toBe(fallbackModel.provider); + expect(session.model?.id).toBe(fallbackModel.id); + expect(fallbackAppliedEvents).toEqual([ + { + type: "retry_fallback_applied", + from: `${primaryModel.provider}/${primaryModel.id}`, + to: `${fallbackModel.provider}/${fallbackModel.id}`, + role: "default", + }, + ]); + }); + it("falls back on structured classifier refusals and pins the fallback", async () => { const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); const fallbackModel = getBundledModel("openai", "gpt-4o-mini"); From 8b17f0d3a17478682ba494892d122a0c9b40e61c Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 4 Jul 2026 17:38:51 +0000 Subject: [PATCH 05/91] fix(coding-agent): scoped TTSR abort reason to matching tool call TTSR stream-interrupt aborts now carry a per-tool reason so the placeholder loop labels only the tool call whose stream matched the rule with the rule name and gives sibling committed tool calls a neutral "TTSR interrupt on another tool call" reason. Previously the single `message.errorMessage` was stamped onto every retained tool-call block, so unrelated read/edit calls read as violating a rule they never matched and misled the model's own reasoning about which call fired. Threads the matched `toolcall:` extracted from the TTSR match context through `agent.abort(...)` as a `ToolScopedAbortReason` object; the agent loop unwraps it in `emitAbortedAssistantMessage` into a `toolCallAbortMessages` map on the aborted `AssistantMessage`, and the `stopReason === "aborted"` fanout in `runAgentLoop` prefers the per-tool message when one exists. Fixes #2783 --- packages/agent/CHANGELOG.md | 4 ++ packages/agent/src/agent-loop.ts | 60 ++++++++++++++++++- packages/ai/CHANGELOG.md | 4 ++ packages/ai/src/types.ts | 2 + packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/session/agent-session.ts | 15 ++++- .../test/agent-session-concurrent.test.ts | 46 +++++++------- 7 files changed, 112 insertions(+), 23 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 26f162c85..d0f642b47 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] + +### Added + +- Added per-tool abort metadata so stream-wide aborts can label matching tool-call placeholders separately from unaffected sibling calls ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.0.1] - 2026-06-15 ### Fixed diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index bfff2aefe..a17798cd1 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -83,6 +83,26 @@ const MAX_PAUSED_TURN_CONTINUATIONS = 8; * tool's own window elapses. A cheap synchronous queue check; latency-bounded * at one tick. */ +/** + * Abort reason for a turn-wide interruption where only some tool calls caused + * the abort and sibling placeholders need neutral messages. + */ +export interface ToolScopedAbortReason { + readonly kind: "tool-scoped-abort"; + readonly message: string; + readonly toolCallMessages: Record; + readonly defaultToolCallMessage: string; +} + +/** Creates an abort reason that labels matching tool calls separately from siblings. */ +export function createToolScopedAbortReason( + message: string, + toolCallMessages: Record, + defaultToolCallMessage: string, +): ToolScopedAbortReason { + return { kind: "tool-scoped-abort", message, toolCallMessages, defaultToolCallMessage }; +} + const STEERING_INTERRUPT_POLL_MS = 250; class HarmonyLeakInterruption extends Error { @@ -158,6 +178,7 @@ function snapshotAssistantMessage(message: AssistantMessage): AssistantMessage { cost: { ...message.usage.cost }, }, disabledFeatures: message.disabledFeatures ? [...message.disabledFeatures] : undefined, + toolCallAbortMessages: message.toolCallAbortMessages ? { ...message.toolCallAbortMessages } : undefined, }; } @@ -761,7 +782,8 @@ async function runLoopBody( const toolCalls = message.content.filter((c): c is ToolCallContent => c.type === "toolCall"); const toolResults: ToolResultMessage[] = []; for (const toolCall of toolCalls) { - const result = createAbortedToolResult(toolCall, stream, message.stopReason, message.errorMessage); + const errorMessage = message.toolCallAbortMessages?.[toolCall.id] ?? message.errorMessage; + const result = createAbortedToolResult(toolCall, stream, message.stopReason, errorMessage); currentContext.messages.push(result); newMessages.push(result); toolResults.push(result); @@ -1392,6 +1414,34 @@ function emitDiscardedHarmonyPartial( }); } +function isStringRecord(value: unknown): value is Record { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + return Object.values(value).every(child => typeof child === "string"); +} + +function toolScopedAbortReason(signal: AbortSignal | undefined): ToolScopedAbortReason | undefined { + const reason = signal?.reason; + if (!reason || typeof reason !== "object") return undefined; + if (Reflect.get(reason, "kind") !== "tool-scoped-abort") return undefined; + if (typeof Reflect.get(reason, "message") !== "string") return undefined; + if (typeof Reflect.get(reason, "defaultToolCallMessage") !== "string") return undefined; + return isStringRecord(Reflect.get(reason, "toolCallMessages")) ? reason : undefined; +} + +function buildToolCallAbortMessages( + message: AssistantMessage, + reason: ToolScopedAbortReason, +): Record | undefined { + let hasToolCall = false; + const messages: Record = {}; + for (const block of message.content) { + if (block.type !== "toolCall") continue; + hasToolCall = true; + messages[block.id] = reason.toolCallMessages[block.id] ?? reason.defaultToolCallMessage; + } + return hasToolCall ? messages : undefined; +} + /** Resolve the human-readable reason an abort carried. A caller that aborts via * `AbortController.abort(reason)` with a string or a non-`AbortError` `Error` * (e.g. the coding agent's user-interrupt label) gets that text surfaced on the @@ -1399,6 +1449,8 @@ function emitDiscardedHarmonyPartial( * `signal.reason` is the default `AbortError` `DOMException`) falls back to the * generic sentinel that downstream renderers treat as "no specific reason". */ export function abortReasonText(signal: AbortSignal | undefined): string { + const scopedReason = toolScopedAbortReason(signal); + if (scopedReason) return scopedReason.message; const reason = signal?.reason; if (typeof reason === "string" && reason.trim().length > 0) return reason; if (reason instanceof Error && reason.name !== "AbortError" && reason.message.trim().length > 0) { @@ -1415,6 +1467,7 @@ export function abortReasonText(signal: AbortSignal | undefined): string { * was observed, so the block is retained and paired with a labeled placeholder; * an anonymous abort drops incomplete calls whose args may be unsafe to replay. */ function isExplicitAbortReason(signal: AbortSignal | undefined): boolean { + if (toolScopedAbortReason(signal)) return true; const reason = signal?.reason; if (typeof reason === "string") return reason.trim().length > 0; if (reason instanceof Error) return reason.name !== "AbortError" && reason.message.trim().length > 0; @@ -1456,6 +1509,11 @@ function emitAbortedAssistantMessage( // `errorMessage`; an anonymous abort still drops calls that never completed // (no `toolcall_end`), whose partial args are unsafe to replay. const retained = isExplicitAbortReason(requestSignal) ? base : retainCompletedToolCalls(base, completedToolCallIds); + const scopedAbort = toolScopedAbortReason(requestSignal); + const toolCallAbortMessages = scopedAbort ? buildToolCallAbortMessages(retained, scopedAbort) : undefined; + if (toolCallAbortMessages) { + retained.toolCallAbortMessages = toolCallAbortMessages; + } const abortedMessage = snapshotAssistantMessage(retained); if (addedPartial) { context.messages[context.messages.length - 1] = abortedMessage; diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index bb43d50d0..50758769b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] + +### Added + +- Added `AssistantMessage.toolCallAbortMessages` for per-tool placeholder labels on aborted assistant turns ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.0.1] - 2026-06-15 ### Added diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 9bd48a5ef..9a55eef0c 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -495,6 +495,8 @@ export interface AssistantMessage { stopReason: StopReason; stopDetails?: StopDetails | null; errorMessage?: string; + /** Per-tool abort messages used when an aborted assistant turn needs different placeholder results per tool call. */ + toolCallAbortMessages?: Record; /** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */ errorStatus?: number; /** diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ce58d520d..6142c5e8b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] + +### Fixed + +- Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.0.1] - 2026-06-15 ### Breaking Changes diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index a5f30688c..3f589a3e9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -31,6 +31,7 @@ import { AppendOnlyContextManager, type AsideMessage, type CompactionSummaryMessage, + createToolScopedAbortReason, resolveTelemetry, STREAM_INTERRUPTED_AFTER_CONTENT_STOP_DETAIL, ThinkingLevel, @@ -2946,7 +2947,8 @@ export class AgentSession { // Decide first: a non-interrupting tool-source match attaches to the // specific tool call's result instead of driving a loop-wide follow-up. const shouldInterrupt = this.#shouldInterruptForTtsrMatch(matches, matchContext); - const perToolId = shouldInterrupt ? undefined : this.#extractTtsrToolCallId(matchContext); + const matchedToolId = this.#extractTtsrToolCallId(matchContext); + const perToolId = shouldInterrupt ? undefined : matchedToolId; if (perToolId) { this.#addPerToolTtsrInjections(perToolId, matches); this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); @@ -2962,7 +2964,16 @@ export class AgentSession { // Abort the stream immediately — do not gate on extension callbacks this.#ttsrAbortPending = true; this.#ensureTtsrResumePromise(); - this.agent.abort(this.#formatTtsrAbortReason(matches)); + const abortReason = this.#formatTtsrAbortReason(matches); + this.agent.abort( + matchedToolId + ? createToolScopedAbortReason( + abortReason, + { [matchedToolId]: abortReason }, + "TTSR interrupt on another tool call", + ) + : abortReason, + ); // Notify extensions (fire-and-forget, does not block abort) this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); // Schedule retry after a short delay diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index cf7504b18..4d31a537b 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -755,7 +755,7 @@ describe("AgentSession TTSR resume gate", () => { expect(session.isStreaming).toBe(false); }); - it("labels aborted tool placeholders with the TTSR rule reason", async () => { + it("labels only the matching aborted tool placeholder with the TTSR rule reason", async () => { collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let streamCallCount = 0; @@ -769,7 +769,13 @@ describe("AgentSession TTSR resume gate", () => { }); ttsrManager.addRule(testRule); - const toolCallContent: ToolCall = { + const readToolCallContent: ToolCall = { + type: "toolCall", + id: "call_innocent_read", + name: "read", + arguments: { path: "history://Eval1WithSkill" }, + }; + const matchedToolCallContent: ToolCall = { type: "toolCall", id: "call_ttsr_abort_reason", name: "mock_edit", @@ -778,7 +784,7 @@ describe("AgentSession TTSR resume gate", () => { const makeToolCallMsg = (stopReason: "toolUse" | "aborted" = "toolUse"): AssistantMessage => ({ role: "assistant", - content: [toolCallContent], + content: [readToolCallContent, matchedToolCallContent], api: "anthropic-messages", provider: "anthropic", model: "mock", @@ -818,10 +824,10 @@ describe("AgentSession TTSR resume gate", () => { ); } stream.push({ type: "start", partial }); - stream.push({ type: "toolcall_start", contentIndex: 0, partial }); + stream.push({ type: "toolcall_start", contentIndex: 1, partial }); stream.push({ type: "toolcall_delta", - contentIndex: 0, + contentIndex: 1, delta: 'let val = result.unwrap("oops")', partial, }); @@ -843,22 +849,22 @@ describe("AgentSession TTSR resume gate", () => { await session.prompt("Write some Rust code"); - const toolResult = sessionManager + const toolResults = sessionManager .getEntries() - .find( - entry => - entry.type === "message" && - entry.message.role === "toolResult" && - entry.message.toolCallId === toolCallContent.id, - ); - expect(toolResult?.type).toBe("message"); - const text = - toolResult?.type === "message" && toolResult.message.role === "toolResult" - ? (toolResult.message.content.find((part): part is { type: "text"; text: string } => part.type === "text") - ?.text ?? "") - : ""; - expect(text).toContain("Tool execution was aborted: TTSR matched rule: no-unwrap"); - expect(text).not.toContain("Request was aborted"); + .filter(entry => entry.type === "message" && entry.message.role === "toolResult") + .map(entry => (entry.type === "message" && entry.message.role === "toolResult" ? entry.message : undefined)) + .filter(message => message !== undefined); + const toolResultText = (toolCallId: string): string => + toolResults + .find(message => message.toolCallId === toolCallId) + ?.content.find((part): part is { type: "text"; text: string } => part.type === "text")?.text ?? ""; + + const readText = toolResultText(readToolCallContent.id); + const matchedText = toolResultText(matchedToolCallContent.id); + expect(readText).toContain("Tool execution was aborted: TTSR interrupt on another tool call"); + expect(readText).not.toContain("TTSR matched rule: no-unwrap"); + expect(matchedText).toContain("Tool execution was aborted: TTSR matched rule: no-unwrap"); + expect(matchedText).not.toContain("Request was aborted"); }); it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { From ce76309065d511f4ec083cf52d6b2589e69c4c39 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 10:06:00 +0000 Subject: [PATCH 06/91] fix(pi-shell): pinned spawn observer identity against Windows pid reuse MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `SpawnRegistry` previously stored only the raw pid reported by brush's `SpawnObserver` hook and deferred `Process::from_pid` to `build_targets` at cancellation time. Between the moment a bash-spawned child exited and the moment cancellation fired, Windows could recycle that freed pid onto an unrelated process — typically another `pwsh.exe` or `powershell.exe` in a different Cursor terminal tab, since PowerShell is the parent shell. `Process::from_pid` at kill time then opened the impostor, and `signal_tree` walked the current Toolhelp snapshot for `ppid == root` matches and `TerminateProcess`'d whatever subtree happened to live under that recycled pid. That is the reporter's Variant A symptom: OMP crashing kills unrelated PowerShell sessions. The `SpawnObserver` impl now pins a stable `Process` handle *at spawn time*, before any pid recycling window can open: - Windows: an open process handle keeps the pid slot reserved for the handle's lifetime (Raymond Chen's documented invariant), so the pid cannot be reassigned while the registry holds a reference. - Linux: the pidfd carries identity independent of the numeric pid. - macOS: the recorded `(pid, start_tvsec, start_tvusec)` triple detects impersonation on every subsequent access. `TerminationTargets::add_process` accepts a pre-pinned handle and skips the `Process::from_pid` re-open entirely, and `build_targets` no longer consults the raw pid at all — an entry the observer failed to pin (the child exited before we could open a handle) becomes a no-op instead of racing pid reuse. Fixes #4605 --- crates/pi-shell/src/process.rs | 137 ++++++++++++++++++++++++++++++--- crates/pi-shell/src/shell.rs | 9 ++- packages/natives/CHANGELOG.md | 4 + 3 files changed, 139 insertions(+), 11 deletions(-) diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs index ede0fd3f8..4eccb80ad 100644 --- a/crates/pi-shell/src/process.rs +++ b/crates/pi-shell/src/process.rs @@ -1571,6 +1571,11 @@ impl TerminationTargets { /// Record a pid. Duplicates are ignored. If the pid is alive, opens /// a stable [`Process`] reference so the descendant tree can be /// killed even if the original pid is reused later. + /// + /// Prefer [`add_process`](Self::add_process) when the caller already holds a + /// [`Process`] captured at spawn time: opening by pid here loses the + /// original identity if the pid was recycled between the child exiting + /// and this call. pub fn add_pid(&mut self, pid: i32) { if self.seen_pids.insert(pid) && let Some(process) = Process::from_pid(pid) @@ -1579,6 +1584,17 @@ impl TerminationTargets { } } + /// Record a pre-pinned [`Process`] handle. Duplicates (by pid) are ignored. + /// + /// This is the correct entry point when the caller captured the handle at + /// spawn time — the handle already pins OS-level identity, so no `from_pid` + /// re-open (and its PID-reuse race) is needed at cancellation time. + pub fn add_process(&mut self, process: Process) { + if self.seen_pids.insert(process.pid()) { + self.processes.push(process); + } + } + /// True when no targets have been recorded. #[must_use] pub const fn is_empty(&self) -> bool { @@ -1599,10 +1615,19 @@ impl TerminationTargets { } /// A single external child reported by the shell's spawn-observer hook. -#[derive(Debug, Clone, Copy)] +/// +/// `process` is captured *at spawn time* so its OS-level identity is pinned +/// before the pid can be recycled. On Windows an open process handle keeps +/// the pid reserved for the lifetime of the reference; on Linux the pidfd +/// pins identity; on macOS the recorded `(pid, start_time)` triple detects +/// impersonation. Storing only the raw pid and re-opening at cancellation +/// time — as previous versions did — leaked kills onto unrelated processes +/// that happened to acquire the recycled pid between the child exiting and +/// the run being cancelled (issue #4605). +#[derive(Clone)] struct SpawnedProcess { - pid: i32, - pgid: Option, + process: Option, + pgid: Option, } /// Per-run record of the OS processes a single shell command launched, @@ -1626,24 +1651,37 @@ impl SpawnRegistry { } /// Record a freshly spawned child. Called from the spawn-observer hook. - pub fn record(&self, pid: i32, pgid: Option) { - self.spawned.lock().push(SpawnedProcess { pid, pgid }); + /// + /// The `Process` handle MUST be opened by the caller *immediately* after + /// the child's pid becomes visible, so identity is pinned before any race + /// with pid recycling can start. When the pin fails (child already exited + /// before we could `Process::from_pid`) the entry becomes a no-op at + /// termination time — there is nothing left to signal. + pub fn record(&self, pgid: Option, process: Option) { + self.spawned.lock().push(SpawnedProcess { process, pgid }); } /// Build the kill set from the processes recorded so far. Re-read on every /// signal wave so a child spawned during a grace window — between the /// cancel firing and the next wave — is still reaped. /// - /// A recorded pid contributes only while alive (`add_pid` opens a stable - /// handle, skipping the dead); a recorded pgid contributes only while the - /// group still has members, so once the run's whole tree exits the targets - /// are empty and the wave loop can stop early. + /// A recorded process contributes only while alive; a recorded pgid + /// contributes only while the group still has members, so once the run's + /// whole tree exits the targets are empty and the wave loop can stop early. #[must_use] pub fn build_targets(&self) -> TerminationTargets { let mut targets = TerminationTargets::new(); let spawned = self.spawned.lock().clone(); for entry in spawned { - targets.add_pid(entry.pid); + if let Some(process) = entry.process { + targets.add_process(process); + } + // If the observer failed to pin a handle at spawn time (the child + // exited before `Process::from_pid` could open it), the child is + // already gone — signalling anything for that pid would either + // no-op or, worse, race a recycled pid onto an unrelated process. + // Drop the entry entirely rather than reintroduce the pid-reuse + // window this whole change exists to close (#4605). if let Some(pgid) = entry.pgid && pgid > 0 && process_group_alive(pgid) @@ -1752,4 +1790,83 @@ mod tests { broken `proc_listchildpids`", ); } + + /// Regression test for issue #4605: `SpawnRegistry` MUST pin a stable + /// [`Process`] reference at spawn time rather than defer re-opening the + /// pid until termination. + /// + /// Before the fix, `SpawnRegistry` stored only the raw pid; `build_targets` + /// called `Process::from_pid` at cancellation time. On Windows pids recycle + /// aggressively, so a bash-spawned `pwsh.exe` that had already exited could + /// see its pid reassigned to an unrelated PowerShell session (e.g. the + /// user's other Cursor terminal). `Process::from_pid` at cancel time would + /// happily open that unrelated process, and `signal_tree` would then + /// enumerate — and `TerminateProcess` — the entire foreign subtree. + /// + /// This test cannot literally trigger Windows pid recycling from a + /// cross-platform Rust test, but it can prove the observable defense: a + /// recorded process reference survives the original pid's death (so no + /// "look it up again" step exists to be raced), and the registry never + /// consults `Process::from_pid` when a handle was pinned at record time. + #[cfg(unix)] + #[test] + fn spawn_registry_pins_identity_at_record_time() { + use std::{process::Command, thread, time::Duration}; + + let mut child = Command::new("sleep") + .arg("30") + .spawn() + .expect("spawn sleep"); + let child_pid = i32::try_from(child.id()).expect("child pid fits in i32"); + + let registry = SpawnRegistry::new(); + // Simulate brush's `on_spawn` hook: pin the handle *now*, while the + // child is definitely alive, before any race with pid recycling could + // start. + let pinned = Process::from_pid(child_pid).expect("pin child at record time"); + registry.record(None, Some(pinned)); + + // Kill the child so the pid becomes eligible for reuse. + let _ = child.kill(); + let _ = child.wait(); + + // Sanity-check that a fresh `Process::from_pid` against the dead pid + // no longer resolves — mirroring the moment on Windows where the pid + // could recycle to an unrelated process before cancellation fires. + for _ in 0..40 { + if Process::from_pid(child_pid).is_none() { + break; + } + thread::sleep(Duration::from_millis(25)); + } + + let targets = registry.build_targets(); + assert!( + !targets.is_empty(), + "SpawnRegistry must retain the pinned handle even after the pid dies; otherwise the \ + kill set is either silently empty (misses legitimate targets) or would need to \ + re-open by pid (racing with pid reuse — issue #4605)" + ); + } + + /// `TerminationTargets::add_process` must accept a pre-pinned handle + /// without going through `Process::from_pid`. This is the API contract + /// `SpawnRegistry` relies on to avoid the PID-reuse race. + #[cfg(unix)] + #[test] + fn add_process_bypasses_from_pid_lookup() { + let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32"); + let pinned = Process::from_pid(self_pid).expect("pin self"); + + let mut targets = TerminationTargets::new(); + targets.add_process(pinned.clone()); + assert!(!targets.is_empty(), "add_process must record the pinned handle"); + + // Adding the same pid again through either entry point must dedupe: + // otherwise every wave in `terminate_run` would re-signal the same + // tree N times. + targets.add_process(pinned); + targets.add_pid(self_pid); + assert_eq!(targets.processes.len(), 1, "duplicate pids must be deduped"); + } } diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index f00982a92..80634bf91 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -1372,7 +1372,14 @@ async fn read_output_bytes( impl SpawnObserver for process::SpawnRegistry { fn on_spawn(&self, pid: i32, pgid: Option) { - self.record(pid, pgid); + // Pin a stable process reference *now*, before the pid can be recycled. + // On Windows an open handle keeps the pid slot reserved for the lifetime + // of the handle; on Linux the pidfd carries identity; on macOS the + // recorded start-time triple detects impersonation. Deferring the open + // to `build_targets` (as the old code did) let a recycled pid resolve + // to an unrelated process — issue #4605. + let process = process::Process::from_pid(pid); + self.record(pgid, process); } } diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index d12104af1..7f029b3a6 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed a Windows regression where an abnormal `omp` exit or bash cancellation could `TerminateProcess` unrelated `pwsh.exe` / `powershell.exe` sessions (including other Cursor terminal tabs). `SpawnRegistry` stored only the raw pid of each brush-spawned child and re-opened it via `Process::from_pid` at cancellation time; between those two moments Windows could recycle a freed pid onto an unrelated PowerShell, and `signal_tree` then walked the wrong subtree via Toolhelp. The observer now pins a stable `Process` handle at spawn time — on Windows the open handle keeps the pid slot reserved, on Linux the pidfd carries identity, on macOS the `(pid, start_time)` triple detects impersonation — so cancellation can only reach children this run actually launched. ([#4605](https://github.com/can1357/oh-my-pi/issues/4605)) + ## [16.3.6] - 2026-07-04 ### Changed From 79e083124c78c1879d3800ab274527f2d42a93d3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 10:14:07 +0000 Subject: [PATCH 07/91] fix(pi-shell): pruned exited entries from SpawnRegistry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the review on #4606: pinning a stable `Process` per spawn is correct against pid reuse, but a long-running shell command that spawns many short-lived external processes (a bash loop invoking a binary per iteration) would otherwise retain one owned OS handle per historical spawn — a pidfd on Linux, a process `HANDLE` on Windows — until the run ends. Under enough iterations that hits the per-process FD/handle limit and starts breaking `Process::from_pid` (or any other file operation) for the rest of the run. `SpawnRegistry` now sweeps entries whose pinned process and process group are both gone. The sweep runs opportunistically inside `record` once the recorded vec crosses a small threshold (`PRUNE_THRESHOLD = 64`) and unconditionally at the top of `build_targets`, so the retained handle count is bounded by the current live tree rather than the historical spawn count. Amortized cost per spawn stays O(1); a sweep is O(N) probes of `Process::status` (non-blocking pidfd `poll` on Linux, `WaitForSingle Object(_, 0)` on Windows), running at most once per `PRUNE_THRESHOLD` records. The previous identity-pinning regression test was rewritten to defend the actual invariant — the pinned handle carries into `build_targets` while the child is alive, and the entry is dropped (never re-opened by pid) once the child exits. A new regression asserts pruning keeps retained entries under `PRUNE_THRESHOLD` after 2×threshold spawns of short-lived children. Fixes #4605 --- crates/pi-shell/src/process.rs | 160 ++++++++++++++++++++++++++++----- packages/natives/CHANGELOG.md | 2 +- 2 files changed, 140 insertions(+), 22 deletions(-) diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs index 4eccb80ad..d9aea6b07 100644 --- a/crates/pi-shell/src/process.rs +++ b/crates/pi-shell/src/process.rs @@ -1644,6 +1644,19 @@ pub struct SpawnRegistry { } impl SpawnRegistry { + /// Amortized-cost threshold for opportunistic pruning of exited entries. + /// + /// A shell run that spawns many short-lived external commands (e.g. a bash + /// loop invoking a binary per iteration) would otherwise retain one owned + /// process handle per spawn — a pidfd on Linux, a `HANDLE` on Windows — for + /// the lifetime of the run, exhausting per-process FD/handle limits. When + /// the recorded vec grows past this many entries, `record` sweeps out + /// entries whose pinned process AND process group are both gone. Cost per + /// sweep is `O(N)` (one non-blocking status probe per entry) and a sweep + /// runs at most once per this many `record` calls, so the amortized cost + /// per spawn is `O(1)` — cheap next to the spawn itself. + const PRUNE_THRESHOLD: usize = 64; + /// Create an empty registry. #[must_use] pub fn new() -> Self { @@ -1657,8 +1670,17 @@ impl SpawnRegistry { /// with pid recycling can start. When the pin fails (child already exited /// before we could `Process::from_pid`) the entry becomes a no-op at /// termination time — there is nothing left to signal. + /// + /// Once the recorded vec reaches [`Self::PRUNE_THRESHOLD`], exited entries + /// are swept so long-running loops of short external commands cannot + /// exhaust the process' FD/handle limit by retaining one owned handle per + /// historical spawn. pub fn record(&self, pgid: Option, process: Option) { - self.spawned.lock().push(SpawnedProcess { process, pgid }); + let mut spawned = self.spawned.lock(); + spawned.push(SpawnedProcess { process, pgid }); + if spawned.len() >= Self::PRUNE_THRESHOLD { + prune_exited(&mut spawned); + } } /// Build the kill set from the processes recorded so far. Re-read on every @@ -1668,10 +1690,17 @@ impl SpawnRegistry { /// A recorded process contributes only while alive; a recorded pgid /// contributes only while the group still has members, so once the run's /// whole tree exits the targets are empty and the wave loop can stop early. + /// + /// Pruning also runs here so a cancellation cycle sees a compact target + /// set even when the record-time threshold hasn't fired yet. #[must_use] pub fn build_targets(&self) -> TerminationTargets { let mut targets = TerminationTargets::new(); - let spawned = self.spawned.lock().clone(); + let spawned = { + let mut guard = self.spawned.lock(); + prune_exited(&mut guard); + guard.clone() + }; for entry in spawned { if let Some(process) = entry.process { targets.add_process(process); @@ -1693,6 +1722,24 @@ impl SpawnRegistry { } } +/// Drop registry entries whose pinned process *and* process group are both +/// gone: with neither still-live, they contribute nothing to the next +/// termination wave and only pin an owned OS handle for no reason. A pgid-only +/// entry (observer failed to pin the process but the group is still alive) is +/// retained so `build_targets` can still signal the group. +fn prune_exited(spawned: &mut Vec) { + spawned.retain(|entry| { + let process_live = entry + .process + .as_ref() + .is_some_and(|process| process.status() == ProcessStatus::Running); + let group_live = entry + .pgid + .is_some_and(|pgid| pgid > 0 && process_group_alive(pgid)); + process_live || group_live + }); +} + /// True when process group `pgid` still has at least one member. `kill(2)` /// with signal 0 performs permission/existence checks without delivering a /// signal; `EPERM` means the group exists but is not ours to signal, which @@ -1813,39 +1860,52 @@ mod tests { fn spawn_registry_pins_identity_at_record_time() { use std::{process::Command, thread, time::Duration}; - let mut child = Command::new("sleep") + // Phase 1: while the child is alive, the pinned handle carries identity + // forward into `build_targets` without any `Process::from_pid` re-open + // step existing to be raced against pid reuse. + let mut long = Command::new("sleep") .arg("30") .spawn() .expect("spawn sleep"); - let child_pid = i32::try_from(child.id()).expect("child pid fits in i32"); + let long_pid = i32::try_from(long.id()).expect("child pid fits in i32"); let registry = SpawnRegistry::new(); - // Simulate brush's `on_spawn` hook: pin the handle *now*, while the - // child is definitely alive, before any race with pid recycling could - // start. - let pinned = Process::from_pid(child_pid).expect("pin child at record time"); + let pinned = Process::from_pid(long_pid).expect("pin child at record time"); registry.record(None, Some(pinned)); - // Kill the child so the pid becomes eligible for reuse. - let _ = child.kill(); - let _ = child.wait(); + let live_targets = registry.build_targets(); + assert!( + !live_targets.is_empty(), + "a still-live pinned child must appear in the target set — otherwise the \ + cancellation cleanup would silently miss it" + ); + let live_pids: Vec = live_targets.processes.iter().map(Process::pid).collect(); + assert_eq!( + live_pids, + vec![long_pid], + "target set must come from the pinned handle recorded at spawn time, not a \ + re-lookup by pid (which would race pid reuse — issue #4605)" + ); - // Sanity-check that a fresh `Process::from_pid` against the dead pid - // no longer resolves — mirroring the moment on Windows where the pid - // could recycle to an unrelated process before cancellation fires. + let _ = long.kill(); + let _ = long.wait(); + + // Phase 2: once the child exits, the registry MUST drop the entry + // rather than reintroduce a `Process::from_pid` re-open at kill time. + // Poll until pruning sees the pidfd as Exited (kernel-visible within + // milliseconds in practice). + let mut empty_after_exit = false; for _ in 0..40 { - if Process::from_pid(child_pid).is_none() { + if registry.build_targets().is_empty() { + empty_after_exit = true; break; } thread::sleep(Duration::from_millis(25)); } - - let targets = registry.build_targets(); assert!( - !targets.is_empty(), - "SpawnRegistry must retain the pinned handle even after the pid dies; otherwise the \ - kill set is either silently empty (misses legitimate targets) or would need to \ - re-open by pid (racing with pid reuse — issue #4605)" + empty_after_exit, + "once the pinned child exits the registry must drop it — re-opening by pid at \ + termination time is exactly the pid-reuse race #4605 closes" ); } @@ -1869,4 +1929,62 @@ mod tests { targets.add_pid(self_pid); assert_eq!(targets.processes.len(), 1, "duplicate pids must be deduped"); } + + /// Regression test for the review on PR #4606: a long-running shell + /// command that spawns many short-lived external processes must not + /// retain one owned handle per historical spawn — that would exhaust + /// per-process FD/handle limits (pidfd on Linux, `HANDLE` on Windows). + /// The registry MUST prune dead entries once the recorded vec crosses + /// the sweep threshold. + #[cfg(unix)] + #[test] + fn spawn_registry_prunes_exited_entries() { + use std::{thread, time::Duration}; + + let registry = SpawnRegistry::new(); + + // Fabricate many recorded-then-exited children by pinning ourselves, + // pushing the entry, then immediately treating it as "dead" from the + // registry's perspective. To simulate the exit without actually + // killing the harness, use `Process::from_pid(1)` for a pid that + // (on Linux) is init and never exits — but wrap the recording in a + // pattern that guarantees `status()` returns Exited for the pruner: + // spawn a tiny child, pin it, wait for exit, then record. + for _ in 0..(SpawnRegistry::PRUNE_THRESHOLD * 2) { + let mut child = std::process::Command::new("true") + .spawn() + .expect("spawn true"); + let pid = i32::try_from(child.id()).expect("child pid fits in i32"); + let pinned = Process::from_pid(pid); + let _ = child.wait(); + // Give the kernel a moment to mark the pidfd readable so `status()` + // reports Exited when the pruner probes. + for _ in 0..20 { + if pinned + .as_ref() + .is_some_and(|process| process.status() == ProcessStatus::Exited) + { + break; + } + thread::sleep(Duration::from_millis(5)); + } + registry.record(None, pinned); + } + + let retained = registry.spawned.lock().len(); + assert!( + retained < SpawnRegistry::PRUNE_THRESHOLD, + "pruning must bound retained entries below the sweep threshold once the pinned \ + processes have exited; got {retained} retained (threshold {})", + SpawnRegistry::PRUNE_THRESHOLD + ); + + // build_targets sees no live handles → empty target set, matching the + // contract that fully-exited registries stop the wave loop early. + let targets = registry.build_targets(); + assert!( + targets.is_empty(), + "registry of only-dead entries must produce an empty target set" + ); + } } diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 7f029b3a6..fdb2fc658 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed a Windows regression where an abnormal `omp` exit or bash cancellation could `TerminateProcess` unrelated `pwsh.exe` / `powershell.exe` sessions (including other Cursor terminal tabs). `SpawnRegistry` stored only the raw pid of each brush-spawned child and re-opened it via `Process::from_pid` at cancellation time; between those two moments Windows could recycle a freed pid onto an unrelated PowerShell, and `signal_tree` then walked the wrong subtree via Toolhelp. The observer now pins a stable `Process` handle at spawn time — on Windows the open handle keeps the pid slot reserved, on Linux the pidfd carries identity, on macOS the `(pid, start_time)` triple detects impersonation — so cancellation can only reach children this run actually launched. ([#4605](https://github.com/can1357/oh-my-pi/issues/4605)) +- Fixed a Windows regression where an abnormal `omp` exit or bash cancellation could `TerminateProcess` unrelated `pwsh.exe` / `powershell.exe` sessions (including other Cursor terminal tabs). `SpawnRegistry` stored only the raw pid of each brush-spawned child and re-opened it via `Process::from_pid` at cancellation time; between those two moments Windows could recycle a freed pid onto an unrelated PowerShell, and `signal_tree` then walked the wrong subtree via Toolhelp. The observer now pins a stable `Process` handle at spawn time — on Windows the open handle keeps the pid slot reserved, on Linux the pidfd carries identity, on macOS the `(pid, start_time)` triple detects impersonation — so cancellation can only reach children this run actually launched. The registry sweeps exited entries once the recorded set crosses a small threshold so a long bash loop of short external commands cannot pin one owned OS handle per historical spawn. ([#4605](https://github.com/can1357/oh-my-pi/issues/4605)) ## [16.3.6] - 2026-07-04 From 1877c7ed2748f228b9d242be2b6ec6e3fd97dd9c Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 10:21:26 +0000 Subject: [PATCH 08/91] fix(pi-shell): kept Windows spawn entries until descendants exit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the second review on #4606: `prune_exited` retained an entry while its process was running OR its process group was alive. On Windows `process_group_alive` is always `false` (no process groups), so the predicate collapsed to "root running" — when a brush-spawned root exited after starting a child that stayed alive (a `pwsh`/shell command that launches a helper and exits, an MCP stdio wrapper handing off to a long-running server), pruning immediately dropped the pinned handle. Dropping that handle closes the last thing keeping the root pid slot reserved. Windows can then recycle the pid onto an unrelated process, reintroducing the exact race #4605 closes. It also strands the leftover child: the next cancellation wave has no root to walk descendants from, so `signal_tree` never reaches it. The retain predicate is now: - root process still running → keep (all platforms); - Windows-only: root exited but the pinned handle still probes at least one live descendant via Toolhelp → keep, because the handle is what guarantees the walk targets the original subtree (recycled pids would not be reachable while the handle holds the slot); - pgid still alive → keep (Unix only; Windows falls through). Unix stays unchanged because a child reparented onto init keeps its pgid, so `process_group_alive` already catches "root gone, descendants alive" without a per-entry tree walk. Fixes #4605 --- crates/pi-shell/src/process.rs | 40 ++++++++++++++++++++++++---------- 1 file changed, 28 insertions(+), 12 deletions(-) diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs index d9aea6b07..ae9ac7e42 100644 --- a/crates/pi-shell/src/process.rs +++ b/crates/pi-shell/src/process.rs @@ -1722,21 +1722,37 @@ impl SpawnRegistry { } } -/// Drop registry entries whose pinned process *and* process group are both -/// gone: with neither still-live, they contribute nothing to the next -/// termination wave and only pin an owned OS handle for no reason. A pgid-only -/// entry (observer failed to pin the process but the group is still alive) is -/// retained so `build_targets` can still signal the group. +/// Drop registry entries whose pinned process, process group, and — on +/// Windows — descendant tree are all gone. With nothing still-live the entry +/// contributes nothing to the next termination wave and only pins an owned OS +/// handle for no reason. +/// +/// The platform split matters because Windows has no process groups. On Unix +/// a child reparented onto init keeps its pgid, so a live pgid still catches +/// grandchildren whose immediate parent exited. On Windows there is no +/// reparenting and no pgid, so we probe the descendant tree directly through +/// the still-open pinned handle — dropping that handle would release the pid +/// slot, letting a recycled pid make future Toolhelp walks unsafe (issue +/// #4605) and orphaning any leftover child from the next cancellation wave. fn prune_exited(spawned: &mut Vec) { spawned.retain(|entry| { - let process_live = entry - .process - .as_ref() - .is_some_and(|process| process.status() == ProcessStatus::Running); - let group_live = entry + if let Some(process) = &entry.process { + if process.status() == ProcessStatus::Running { + return true; + } + // Windows-only: root exited but the pinned handle still keeps its + // pid reserved, so `live_descendants` walks the *original* subtree + // via Toolhelp. If any child is still running we must keep the + // entry — closing the handle would both release the pid (racing + // pid reuse) and strand the surviving child. + #[cfg(target_os = "windows")] + if !process.live_descendants().is_empty() { + return true; + } + } + entry .pgid - .is_some_and(|pgid| pgid > 0 && process_group_alive(pgid)); - process_live || group_live + .is_some_and(|pgid| pgid > 0 && process_group_alive(pgid)) }); } From 04258b98d7a67fb7976f48b4511967cc0ea92f43 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 10:27:39 +0000 Subject: [PATCH 09/91] fix(pi-shell): bounded SpawnRegistry sweeps by watermark not raw length MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the third review on #4606: `record` guarded pruning with `spawned.len() >= PRUNE_THRESHOLD`. Once the recorded vec stabilized above the threshold with entries the sweep could not remove — a run whose live children exceed the threshold, e.g. `for i in {1..1000}; do sleep 60 & done` — every subsequent `record` re-entered `prune_exited` while holding the registry lock. Each sweep is O(N) (a status probe per entry, plus a Toolhelp descendant walk on Windows for exited roots), so the per-spawn cost climbed to O(n²) even though the doc-comment promised amortized O(1). The registry now tracks a `next_sweep_at` watermark alongside the recorded vec (both under one mutex — the two fields are always mutated together). A sweep fires only when `spawned.len()` crosses that watermark; every sweep rescheds the next fire `PRUNE_THRESHOLD` further records away from the current post-sweep size. Sweep frequency is now capped at one per `PRUNE_THRESHOLD` records regardless of live set size, restoring true amortized O(1) per spawn. `build_targets` resets the watermark after its own sweep so the record-time schedule stays consistent with the post-cancel live set. Added `spawn_registry_watermark_bounds_sweep_frequency`: fills the vec past threshold with permanently-live entries, records another 20, and asserts the vec grew by exactly 20 (no sweep modified it) and the watermark did not advance. Existing tests updated for the collapsed `state` mutex. Fixes #4605 --- crates/pi-shell/src/process.rs | 119 +++++++++++++++++++++++++++------ 1 file changed, 100 insertions(+), 19 deletions(-) diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs index ae9ac7e42..67a78f187 100644 --- a/crates/pi-shell/src/process.rs +++ b/crates/pi-shell/src/process.rs @@ -1638,9 +1638,24 @@ struct SpawnedProcess { /// host process: a run that cancelled would signal *any* descendant spawned /// after its baseline, including another run's children. Ownership is now /// explicit — only processes this run actually spawned are ever signalled. +#[derive(Default)] +struct RegistryState { + spawned: Vec, + /// The next `spawned.len()` at which `record` runs a sweep. Bounds sweep + /// frequency when the live set stabilizes above the initial threshold: + /// without this watermark, every subsequent `record` would find + /// `len >= PRUNE_THRESHOLD` true and sweep on every spawn (O(n²) in a + /// large-fan-out run like `for i in {1..1000}; do sleep 60 & done`). With + /// it, the next sweep only fires once the vec has grown by another + /// `PRUNE_THRESHOLD` entries since the previous sweep — restoring true + /// amortized O(1) per spawn regardless of how many entries survive each + /// sweep. + next_sweep_at: usize, +} + #[derive(Default)] pub struct SpawnRegistry { - spawned: Mutex>, + state: Mutex, } impl SpawnRegistry { @@ -1649,12 +1664,15 @@ impl SpawnRegistry { /// A shell run that spawns many short-lived external commands (e.g. a bash /// loop invoking a binary per iteration) would otherwise retain one owned /// process handle per spawn — a pidfd on Linux, a `HANDLE` on Windows — for - /// the lifetime of the run, exhausting per-process FD/handle limits. When - /// the recorded vec grows past this many entries, `record` sweeps out - /// entries whose pinned process AND process group are both gone. Cost per - /// sweep is `O(N)` (one non-blocking status probe per entry) and a sweep - /// runs at most once per this many `record` calls, so the amortized cost - /// per spawn is `O(1)` — cheap next to the spawn itself. + /// the lifetime of the run, exhausting per-process FD/handle limits. + /// + /// Each sweep costs `O(N)` (one non-blocking status probe per entry, plus + /// a Toolhelp descendant walk on Windows for exited roots). The next sweep + /// is scheduled `PRUNE_THRESHOLD` further records away — via the + /// `next_sweep_at` watermark — so a run that keeps many concurrent + /// long-lived children (`for i in {1..1000}; do sleep 60 & done`) does not + /// sweep on every spawn just because the vec is already above threshold. + /// Amortized cost per spawn stays `O(1)` regardless of the live-set size. const PRUNE_THRESHOLD: usize = 64; /// Create an empty registry. @@ -1671,15 +1689,22 @@ impl SpawnRegistry { /// before we could `Process::from_pid`) the entry becomes a no-op at /// termination time — there is nothing left to signal. /// - /// Once the recorded vec reaches [`Self::PRUNE_THRESHOLD`], exited entries - /// are swept so long-running loops of short external commands cannot - /// exhaust the process' FD/handle limit by retaining one owned handle per - /// historical spawn. + /// Exited entries are swept opportunistically once the recorded vec + /// crosses the next-sweep watermark, so long-running loops of short + /// external commands cannot exhaust the process' FD/handle limit by + /// retaining one owned handle per historical spawn. pub fn record(&self, pgid: Option, process: Option) { - let mut spawned = self.spawned.lock(); - spawned.push(SpawnedProcess { process, pgid }); - if spawned.len() >= Self::PRUNE_THRESHOLD { - prune_exited(&mut spawned); + let mut state = self.state.lock(); + state.spawned.push(SpawnedProcess { process, pgid }); + if state.spawned.len() >= state.next_sweep_at.max(Self::PRUNE_THRESHOLD) { + prune_exited(&mut state.spawned); + // Schedule the next sweep `PRUNE_THRESHOLD` further records away. + // Comparing against the post-sweep live-set size (not the pre-sweep + // length) bounds the sweep frequency when many entries survive: + // each sweep costs O(N) but now runs at most once per + // `PRUNE_THRESHOLD` records, so amortized per-record cost is O(1) + // even if the live set stays large. + state.next_sweep_at = state.spawned.len() + Self::PRUNE_THRESHOLD; } } @@ -1697,9 +1722,13 @@ impl SpawnRegistry { pub fn build_targets(&self) -> TerminationTargets { let mut targets = TerminationTargets::new(); let spawned = { - let mut guard = self.spawned.lock(); - prune_exited(&mut guard); - guard.clone() + let mut state = self.state.lock(); + prune_exited(&mut state.spawned); + // Reset the watermark to the current live-set size + threshold; + // leaving a stale pre-sweep value would misgate the next + // record-time sweep. + state.next_sweep_at = state.spawned.len() + Self::PRUNE_THRESHOLD; + state.spawned.clone() }; for entry in spawned { if let Some(process) = entry.process { @@ -1987,7 +2016,7 @@ mod tests { registry.record(None, pinned); } - let retained = registry.spawned.lock().len(); + let retained = registry.state.lock().spawned.len(); assert!( retained < SpawnRegistry::PRUNE_THRESHOLD, "pruning must bound retained entries below the sweep threshold once the pinned \ @@ -2003,4 +2032,56 @@ mod tests { "registry of only-dead entries must produce an empty target set" ); } + + /// Regression test for the third review on PR #4606: once the recorded + /// vec crosses `PRUNE_THRESHOLD`, subsequent `record` calls must NOT + /// sweep on every spawn. Without the `next_sweep_at` watermark, a large + /// fan-out run whose live children exceed the threshold turned every + /// spawn into an O(N) status probe of the whole retained set. + /// + /// The check reasons about the observable side effect: after N records + /// past threshold with entries that CANNOT be pruned (all still live), + /// the retained size grows monotonically by exactly N — no sweep runs + /// have modified the vec in between. The direct signal of "did a sweep + /// happen" is a stable pinned handle count across records. + #[cfg(unix)] + #[test] + fn spawn_registry_watermark_bounds_sweep_frequency() { + let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32"); + let registry = SpawnRegistry::new(); + + // Fill past threshold with entries that are permanently alive + // (pinning ourselves) so the pruner has nothing to remove. + let fill = SpawnRegistry::PRUNE_THRESHOLD + 10; + for _ in 0..fill { + registry.record(None, Process::from_pid(self_pid)); + } + let after_fill = registry.state.lock().spawned.len(); + assert_eq!( + after_fill, fill, + "live-only entries must not be pruned during warm-up" + ); + let watermark_after_fill = registry.state.lock().next_sweep_at; + + // Every additional record with a live entry must land in the vec + // verbatim and — critically — NOT re-enter `prune_exited` until the + // vec crosses the freshly scheduled watermark. If the guard were + // still `len >= PRUNE_THRESHOLD` (pre-fix), a sweep would fire on + // every one of these records. + let extra = 20; + for _ in 0..extra { + registry.record(None, Process::from_pid(self_pid)); + } + let after_extra = registry.state.lock().spawned.len(); + assert_eq!( + after_extra, + after_fill + extra, + "records with live entries must accumulate without triggering per-spawn sweeps" + ); + assert_eq!( + registry.state.lock().next_sweep_at, + watermark_after_fill, + "watermark must not advance while the vec stays below it — otherwise a sweep ran" + ); + } } From 1055d4e0d5a9f9694231a04cf8ed774bffb412d7 Mon Sep 17 00:00:00 2001 From: chan1103 Date: Sun, 5 Jul 2026 19:17:03 +0900 Subject: [PATCH 10/91] fix(tui): hang wrapped box-drawing tree lines under their node text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prose lines shaped like tree guides ("├── item") wrapped by restarting at column 0, shearing the tree apart — doubly fast for CJK text. Wrap the node text within the cells remaining after the guide prefix and re-emit the prefix on continuation rows in pass-through form (├ → │, └ → blank), so ancestor rails stay joined. Paragraph/text blocks only; fitting lines, non-tree prose, and code blocks are byte-for-byte unchanged. --- packages/tui/CHANGELOG.md | 1 + packages/tui/src/components/markdown.ts | 151 ++++++++- packages/tui/test/markdown-tree-wrap.test.ts | 318 +++++++++++++++++++ 3 files changed, 469 insertions(+), 1 deletion(-) create mode 100644 packages/tui/test/markdown-tree-wrap.test.ts diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5d2691df2..0dcc9b833 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed submitted slash-command arguments treating `@` file-reference tokens as prompt-composer autocomplete triggers when the command does not define argument completions. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) +- Fixed box-drawing tree lines (`├── item` — directory layouts, decision trees) in prose shearing apart when they wrap: continuation rows now hang under the node text with ancestor rails carried through (`├` → `│`, `└` → blank) instead of restarting at column 0. Applies to prose paragraphs (including inside blockquotes) only when a line with a branch-connector prefix (`├──`, `└─`, …) actually overflows; fitting lines, non-tree prose, and code blocks render byte-for-byte as before. ## [16.3.6] - 2026-07-04 diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index b6eee348c..e81cd52aa 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -251,6 +251,155 @@ function splitTerminalLines(text: string): string[] { return lines; } +// --------------------------------------------------------------------------- +// Tree-guide hanging wrap +// +// Models routinely emit box-drawing trees ("├── item") inside plain +// paragraphs — directory layouts, decision trees. The lexer sees those lines +// as ordinary prose, so the generic wrap pass restarts wrapped continuations +// at column 0 and visually shears the tree apart (doubly fast for CJK text, +// where every glyph is two cells wide). Mirror the guide semantics of +// `tree(1)` / rich.tree instead: wrap the node text within the cells that +// remain after the guide prefix, and indent every continuation row under the +// node text — branch glyphs swap to their pass-through form (`├` → `│`, +// `└` → blank) so the rails of still-open ancestors stay visually joined. +// --------------------------------------------------------------------------- + +/** Continuation glyph for each guide character a tree prefix may contain. */ +const TREE_GUIDE_CONTINUATION: Record = { + "│": "│", + "┃": "┃", + "║": "║", + "├": "│", + "┣": "┃", + "╠": "║", + "└": " ", + "┗": " ", + "╚": " ", + "╰": " ", + "─": " ", + "━": " ", + "═": " ", + " ": " ", +}; + +/** Cheap pre-gate: any guide glyph at all. The structural test is TREE_BRANCH_CONNECTOR_RE. */ +const TREE_GUIDE_ANCHOR_RE = /[│┃║├┣╠└┗╚╰]/; + +/** + * A prefix qualifies as tree-shaped only when a branch/corner glyph is + * immediately followed by a horizontal connector (`├──`, `└─`, `╰──`, …). + * A lone rail or branch glyph used as prose ("│ is the Unicode vertical box + * drawing glyph…") never qualifies, so such paragraphs keep the plain wrap. + */ +const TREE_BRANCH_CONNECTOR_RE = /[├┣╠└┗╚╰][─━═]/; + +/** Below this many content cells a hanging wrap degenerates; keep the plain wrap. */ +const MIN_TREE_CONTENT_WIDTH = 8; + +const SGR_SEQUENCE_STICKY = /\x1b\[[0-9;:]*m/y; +const SGR_SEQUENCE_GLOBAL = /\x1b\[[0-9;:]*m/g; + +/** + * Everything before the last full SGR reset is dead state — drop it so the + * re-played `carry` stays bounded by the paragraph's live style run instead + * of its whole code history. + */ +function compactSgrCarry(carry: string): string { + const shortReset = carry.lastIndexOf("\x1b[m"); + const longReset = carry.lastIndexOf("\x1b[0m"); + const cut = Math.max(shortReset === -1 ? -1 : shortReset + 3, longReset === -1 ? -1 : longReset + 4); + return cut === -1 ? carry : carry.slice(cut); +} + +interface TreeGuidePrefix { + /** Index of the first char past the guide run (start of the node text). */ + end: number; + /** SGR sequences interleaved with the guides, in order (zero visible width). */ + codes: string; + /** Guide characters with SGR stripped, exactly as they appear on screen. */ + guides: string; +} + +/** + * Match the leading box-drawing guide run of a rendered line (e.g. `│ ├── `), + * tolerating interleaved SGR styling. Returns undefined unless the run + * contains a branch glyph joined to a horizontal connector and node text + * follows, so dash art, indented prose, and lone glyphs used as prose are + * never treated as a tree. + */ +function matchTreeGuidePrefix(line: string): TreeGuidePrefix | undefined { + let codes = ""; + let guides = ""; + let i = 0; + while (i < line.length) { + if (line.charCodeAt(i) === 0x1b) { + SGR_SEQUENCE_STICKY.lastIndex = i; + const match = SGR_SEQUENCE_STICKY.exec(line); + if (!match) break; + codes += match[0]; + i = SGR_SEQUENCE_STICKY.lastIndex; + continue; + } + const char = line[i]!; + if (!(char in TREE_GUIDE_CONTINUATION)) break; + guides += char; + i++; + } + if (i >= line.length || !TREE_BRANCH_CONNECTOR_RE.test(guides)) return undefined; + return { end: i, codes, guides }; +} + +/** + * Hanging wrap for box-drawing tree lines inside prose block text. + * + * Returns undefined when no line needs the treatment, so paragraphs without + * overflowing tree lines keep their exact current render. When a paragraph + * does hang, its lines are returned pre-split and style-self-contained: the + * SGR state open at each line start is re-played onto that line (`carry`), + * because the caller's wrap pass — which normally carries SGR state across + * the newlines of a single entry — no longer sees them as one entry. + */ +function hangWrapTreeGuideLines(text: string, width: number): string[] | undefined { + if (width < MIN_TREE_CONTENT_WIDTH || !TREE_GUIDE_ANCHOR_RE.test(text)) return undefined; + + const sourceLines = text.split("\n"); + const hangs = (line: string): TreeGuidePrefix | undefined => { + if (visibleWidth(line) <= width) return undefined; + const prefix = matchTreeGuidePrefix(line); + if (!prefix) return undefined; + if (width - visibleWidth(prefix.guides) < MIN_TREE_CONTENT_WIDTH) return undefined; + return prefix; + }; + if (!sourceLines.some(line => hangs(line) !== undefined)) return undefined; + + const out: string[] = []; + let carry = ""; + for (const line of sourceLines) { + const prefix = hangs(line); + if (!prefix) { + out.push(carry ? carry + line : line); + carry = compactSgrCarry(carry + (line.match(SGR_SEQUENCE_GLOBAL)?.join("") ?? "")); + continue; + } + // Re-play the SGR state ahead of the node text so the wrapper carries + // it onto every continuation row; the codes are zero-width, so measured + // row widths are unaffected. + const activeCodes = carry + prefix.codes; + const rows = wrapTextWithAnsi(activeCodes + line.slice(prefix.end), width - visibleWidth(prefix.guides)); + let hang = ""; + for (const guide of prefix.guides) hang += TREE_GUIDE_CONTINUATION[guide] ?? " "; + const hangShortfall = visibleWidth(prefix.guides) - visibleWidth(hang); + if (hangShortfall > 0) hang += padding(hangShortfall); + out.push(carry + line.slice(0, prefix.end) + rows[0]!.slice(activeCodes.length)); + for (let i = 1; i < rows.length; i++) { + out.push(activeCodes + hang + rows[i]!); + } + carry = compactSgrCarry(carry + (line.match(SGR_SEQUENCE_GLOBAL)?.join("") ?? "")); + } + return out; +} + class StrictStrikethroughTokenizer extends Tokenizer { override del(src: string): Tokens.Del | undefined { const match = STRICT_STRIKETHROUGH_REGEX.exec(src); @@ -1383,7 +1532,7 @@ export class Markdown implements Component { break; } const paragraphText = this.#renderInlineTokens(token.tokens || [], styleContext); - lines.push(paragraphText); + lines.push(...(hangWrapTreeGuideLines(paragraphText, width) ?? [paragraphText])); // Don't add spacing if next token is space or list if (nextTokenType && nextTokenType !== "list" && nextTokenType !== "space") { lines.push(""); diff --git a/packages/tui/test/markdown-tree-wrap.test.ts b/packages/tui/test/markdown-tree-wrap.test.ts new file mode 100644 index 000000000..16097efcc --- /dev/null +++ b/packages/tui/test/markdown-tree-wrap.test.ts @@ -0,0 +1,318 @@ +import { describe, expect, it } from "bun:test"; +import { stripVTControlCharacters } from "node:util"; +import { Markdown } from "@oh-my-pi/pi-tui/components/markdown"; +import { visibleWidth, wrapTextWithAnsi } from "@oh-my-pi/pi-tui/utils"; +import { Chalk } from "chalk"; +import { defaultMarkdownTheme } from "./test-themes.js"; + +const WIDTH = 40; + +function renderRaw(text: string, width = WIDTH): readonly string[] { + return new Markdown(text, 0, 0, defaultMarkdownTheme).render(width); +} + +/** Rendered rows as plain text, right-padding stripped (rows are padded to full width). */ +function renderPlain(text: string, width = WIDTH): string[] { + return renderRaw(text, width).map(line => stripVTControlCharacters(line).trimEnd()); +} + +describe("Markdown tree-guide hanging wrap", () => { + it("hangs an overflowing '├── ' node under the node-text column with double-width Korean text", () => { + const node = "가나다라 마바사아 자차카타 파하가나 다라마바 사자차카"; // 6 words x 8 cells + const raw = renderRaw(`├── ${node}`); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("├── 가나다라")).toBeTruthy(); + + for (const line of plain.slice(1)) { + // Exactly `│` + 3 spaces: the node text column is cell 4, so the + // continuation text must begin right there — not a cell earlier or later. + expect(line.startsWith("│ ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + } + for (const line of raw) { + expect(visibleWidth(line)).toBeLessThanOrEqual(WIDTH); + } + + // No glyph may be lost or duplicated by the wrap (spaces are consumed at + // break points, so compare with spaces removed). + const rejoined = [plain[0]!.slice("├── ".length), ...plain.slice(1).map(line => line.slice("│ ".length))] + .join("") + .replace(/ /g, ""); + expect(rejoined).toBe(node.replace(/ /g, "")); + }); + + it("keeps the outer rail and releases the corner for a nested '│ └── ' node", () => { + const plain = renderPlain("│ └── delta echo foxtrot golf hotel india juliet"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("│ └── delta")).toBeTruthy(); + for (const line of plain.slice(1)) { + // `│` stays (outer level still open), `└──` releases to spaces. + expect(line.startsWith("│ ")).toBeTruthy(); + expect(line[8]).not.toBe(" "); + } + }); + + it("releases a last-child '└── ' node to pure spaces with no rail on continuations", () => { + const plain = renderPlain("└── alpha bravo charlie delta echo foxtrot golf hotel"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("└── alpha")).toBeTruthy(); + for (const line of plain.slice(1)) { + expect(line.startsWith(" ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + expect(line.includes("│")).toBeFalsy(); + } + }); + + it("treats the rounded corner '╰── ' like '└── '", () => { + const plain = renderPlain("╰── alpha bravo charlie delta echo foxtrot golf hotel"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("╰── alpha")).toBeTruthy(); + for (const line of plain.slice(1)) { + expect(line.startsWith(" ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + expect(line.includes("│")).toBeFalsy(); + } + }); + + it("leaves a non-tree paragraph byte-identical to the plain wrap", () => { + const text = "alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima"; + const plain = renderPlain(text); + + // Differential against the generic wrapper: the tree feature must not + // have touched this paragraph at all. + const expected = wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd()); + expect(plain).toEqual(expected); + + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("does not treat a dash-only '── ' start as a tree", () => { + const text = "── alpha bravo charlie delta echo foxtrot golf hotel india"; + const plain = renderPlain(text); + + const expected = wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd()); + expect(plain).toEqual(expected); + + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + // Flush at column 0: no injected hang, no rail. + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("renders a fitting tree line-for-line unchanged", () => { + const plain = renderPlain("├── alpha\n│ └── beta\n└── gamma"); + + expect(plain).toEqual(["├── alpha", "│ └── beta", "└── gamma"]); + }); + + it("keeps the old column-0 wrap for '├── ' lines inside fenced code blocks", () => { + const codeLine = "├── alpha bravo charlie delta echo foxtrot golf hotel india"; + const raw = renderRaw(`\`\`\`\n${codeLine}\n\`\`\``); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain[0]).toBe("```"); + expect(plain[plain.length - 1]).toBe("```"); + + const treeRow = plain.findIndex(line => line.includes("├──")); + expect(treeRow).toBeGreaterThan(0); + // The code line overflows, so a continuation row exists before the + // closing fence — and it starts flush at column 0, no hanging prefix. + const continuation = plain[treeRow + 1]!; + expect(treeRow + 1).toBeLessThan(plain.length - 1); + expect(continuation.length).toBeGreaterThan(0); + expect(continuation[0]).not.toBe(" "); + expect(continuation[0]).not.toBe("│"); + + for (const line of raw) { + expect(visibleWidth(line)).toBeLessThanOrEqual(WIDTH); + } + }); + + it("carries an open bold span onto the continuation row", () => { + const raw = renderRaw("├── aaaa bbbb **cccc dddd eeee ffff gggg hhhh**"); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + // Row 0 holds 36 node cells ("aaaa bbbb cccc dddd eeee ffff gggg"), + // so "hhhh" — inside the bold span — lands on the continuation row. + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[1]!.startsWith("│ hhhh")).toBeTruthy(); + + const continuation = raw[1]!; + const boldOpen = continuation.indexOf("\x1b[1m"); + expect(boldOpen).toBeGreaterThanOrEqual(0); + expect(boldOpen).toBeLessThan(continuation.indexOf("hhhh")); + }); + + it("hangs inside a blockquote, after the quote border", () => { + const plain = renderPlain("> ├── alpha bravo charlie delta echo foxtrot golf hotel india"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + // Quote border symbol, border gap, then the tree prefix. + expect(plain[0]!.startsWith("│ ├── alpha")).toBeTruthy(); + const continuations = plain.slice(1).filter(line => line !== ""); // drop trailing spacing rows + expect(continuations.length).toBeGreaterThanOrEqual(1); + for (const line of continuations) { + expect(line.startsWith("│ │ ")).toBeTruthy(); + expect(line[6]).not.toBe(" "); + } + }); + + describe("detection strictness and carry edge cases", () => { + const SGR_RE = /\x1b\[[0-9;:]*m/g; + + it("keeps plain wrap for prose starting with a lone '│ ' rail glyph", () => { + const text = "│ is the Unicode vertical box drawing glyph used for rails in terminal trees"; + const raw = renderRaw(text); + + // Byte-identical to the generic wrapper: no branch+connector pair, + // so the tree feature must not inject rails or indent. + expect(raw.map(line => line.trimEnd())).toEqual(wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd())); + + const plain = renderPlain(text); + expect(plain.length).toBeGreaterThanOrEqual(2); // the paragraph really overflowed + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("keeps plain wrap for prose starting with '├ ' without a horizontal connector", () => { + const text = "├ marks a branch point in a tree diagram and has no horizontal connector here"; + const raw = renderRaw(text); + + expect(raw.map(line => line.trimEnd())).toEqual(wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd())); + + const plain = renderPlain(text); + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("falls back to plain wrap when fewer than 8 content cells remain after the prefix", () => { + const text = "├── alpha bravo charlie delta echo foxtrot golf hotel india"; + + // Width 10 leaves 10 - 4 = 6 content cells after the '├── ' prefix: + // below the minimum, so the hang degenerates and plain wrap wins. + const raw = renderRaw(text, 10); + expect(raw.map(line => line.trimEnd())).toEqual(wrapTextWithAnsi(text, 10).map(line => line.trimEnd())); + const plain = renderPlain(text, 10); + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + + // Width 12 leaves exactly 8 content cells — the boundary where the + // hanging wrap applies again. + const hung = renderPlain(text, 12); + expect(hung[0]).toBe("├── alpha"); + expect(hung.length).toBeGreaterThanOrEqual(2); + for (const line of hung.slice(1)) { + expect(line.startsWith("│ ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + } + }); + + it("replays SGR state opened on an earlier line onto a later hung line's continuation rows", () => { + // With a default text style, the renderer re-opens the default color + // after `**bold**` (the style prefix) and the soft break leaves that + // re-open unclosed — a style opened on line 1 that is still active + // when line 2 hangs. Line 1's bold open/close pair exists nowhere on + // line 2, so finding it ahead of the continuation text proves the + // carry was re-played rather than line 2's own codes. + const chalk = new Chalk({ level: 3 }); + const raw = new Markdown( + "aaa **bold**\n├── alpha bravo charlie delta echo foxtrot golf hotel india", + 0, + 0, + defaultMarkdownTheme, + { color: text => chalk.red(text) }, + ).render(WIDTH); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain.length).toBe(3); + expect(plain[1]!.startsWith("├── alpha")).toBeTruthy(); + expect(plain[2]!.startsWith("│ foxtrot")).toBeTruthy(); + + // Line 1 genuinely ends with an unclosed style: its last SGR is the + // default-color re-open, not a close. + const row0Codes = raw[0]!.trimEnd().match(SGR_RE)!; + expect(row0Codes[row0Codes.length - 1]).toBe("\x1b[31m"); + + // The continuation row starts with a zero-width SGR run that replays + // line 1's history (the carried bold pair) and nets out to the + // default color being open ahead of the visible text. + const continuation = raw[2]!; + const hangAt = continuation.indexOf("│ "); + expect(hangAt).toBeGreaterThan(0); + const replayed = continuation.slice(0, hangAt); + expect(replayed.replace(SGR_RE, "")).toBe(""); + expect(replayed).toContain("\x1b[1m"); + const replayedCodes = replayed.match(SGR_RE)!; + expect(replayedCodes[replayedCodes.length - 1]).toBe("\x1b[31m"); + }); + + it("re-renders byte-identically after a width round-trip (44 → 80 → 44)", () => { + const doc = "├── alpha bravo charlie delta echo foxtrot golf\n└── hotel india juliet kilo lima mike november"; + const md = new Markdown(doc, 0, 0, defaultMarkdownTheme); + + const first = [...md.render(44)]; + // Narrow render actually hung — the round-trip below is not vacuous. + expect(first.map(line => stripVTControlCharacters(line).trimEnd())).toEqual([ + "├── alpha bravo charlie delta echo foxtrot", + "│ golf", + "└── hotel india juliet kilo lima mike", + " november", + ]); + + // Wide render fits line-for-line: a genuinely different layout. + const wide = md.render(80).map(line => stripVTControlCharacters(line).trimEnd()); + expect(wide).toEqual([ + "├── alpha bravo charlie delta echo foxtrot golf", + "└── hotel india juliet kilo lima mike november", + ]); + + expect([...md.render(44)]).toEqual(first); + expect([...new Markdown(doc, 0, 0, defaultMarkdownTheme).render(44)]).toEqual(first); + }); + + it("does not leak styles opened before a full SGR reset onto later hung rows", () => { + // Raw ANSI in component input passes through marked byte-exact: line 1 + // opens bold, fully resets, then opens italic. Only the italic — the + // live style after the reset — may carry onto the hung line. + const raw = renderRaw( + "aaa \x1b[1mbold\x1b[0m\x1b[3mrest and filler\n├── alpha bravo charlie delta echo foxtrot golf hotel india", + ); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain.length).toBe(3); + expect(plain[1]!.startsWith("├── alpha")).toBeTruthy(); + expect(plain[2]!.startsWith("│ foxtrot")).toBeTruthy(); + + for (const row of raw.slice(1)) { + expect(row).not.toContain("\x1b[1m"); // dead pre-reset style must not re-play + expect(row).not.toContain("\x1b[0m"); + } + // The live post-reset style carries onto the hung line and is the + // entire replayed run ahead of the continuation's hang glyphs. + expect(raw[1]!.startsWith("\x1b[3m├── ")).toBeTruthy(); + const continuation = raw[2]!; + const hangAt = continuation.indexOf("│ "); + expect(hangAt).toBeGreaterThan(0); + expect(continuation.slice(0, hangAt)).toBe("\x1b[3m"); + }); + }); +}); From d9ef874e2b143316a73230c160d5a1420e579746 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 12:02:47 +0000 Subject: [PATCH 11/91] fix(discovery): accept array-form Claude plugin manifest paths MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Claude plugin manifest allows `commands`, `slash-commands`, and `skills` to be either a single string or an array of strings — documented under https://code.claude.com/docs/en/plugins-reference#path-behavior-rules and used by real marketplace plugins such as addyosmani/agent-skills whose plugin.json declares `"commands": ["./.claude/commands", "./commands"]`. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those manifest fields as `string` only, so array-shaped values were silently dropped: no items loaded, no warning surfaced. Slash commands (`spec`, `plan`, `build`, `test`, `review`) never appeared in the picker. Normalize `string | string[]` at the resolver, load every in-root entry, and emit one out-of-plugin-root warning per bad entry so misconfigured paths remain observable. Skills and slash-command loaders now fan out over the resolved directory list and merge warnings from all sources. Fixes #4609 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/discovery/claude-plugins.ts | 137 ++++++++++++------ .../test/discovery/claude-plugins.test.ts | 126 ++++++++++++++++ 3 files changed, 218 insertions(+), 46 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ebe03ca8d..2d7ad6da5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry so misconfigured paths remain visible. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609)) - Fixed macOS Backspace on empty search not deleting sessions in the `/resume` picker; Fn+Backspace terminals that deliver `\x7f` instead of `\e[3~` now reach the delete confirmation dialog. ([#4580](https://github.com/can1357/oh-my-pi/pull/4580) by [@JagravNaik](https://github.com/JagravNaik)) - Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal session title text. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) - Fixed empty session `.jsonl` files accumulating in `~/.omp/agent/sessions//` after a draft-then-clear exit cycle. `SessionManager.saveDraft(text)` materializes the session file so the draft sidecar has a parent; a subsequent `saveDraft("")` unlinked the sidecar but left the metadata-only JSONL behind (title slot + session header + startup selector entries, ~500–750 B), and `#shouldHaveSessionFile()` could no longer prune it once `#fileIsCurrent`/`#forceFileCreation` were latched. `SessionManager.close()` now drops only draft-owned metadata-only sessions with no saved draft sidecar to reattach to, while keeping real conversations, meaningful non-message entries such as handoff custom messages, explicit `ensureOnDisk()` sessions, drafts still pending for `--resume`, and never-materialized sessions untouched ([#4571](https://github.com/can1357/oh-my-pi/issues/4571)). diff --git a/packages/coding-agent/src/discovery/claude-plugins.ts b/packages/coding-agent/src/discovery/claude-plugins.ts index c980eec98..8c46b7112 100644 --- a/packages/coding-agent/src/discovery/claude-plugins.ts +++ b/packages/coding-agent/src/discovery/claude-plugins.ts @@ -30,14 +30,14 @@ const DISPLAY_NAME = "Claude Code Marketplace"; const PRIORITY = 70; // Below claude.ts (80) so user .claude/ overrides win interface ClaudePluginManifest { - skills?: string; - "slash-commands"?: string; - commands?: string; + skills?: string | string[]; + "slash-commands"?: string | string[]; + commands?: string | string[]; } interface ResolvedPluginDir { - dir: string; - warning?: string; + dirs: string[]; + warnings: string[]; } async function readPluginManifest(root: ClaudePluginRoot): Promise { @@ -59,6 +59,19 @@ function isWithinPluginRoot(rootPath: string, targetPath: string): boolean { return relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative)); } +/** + * Resolve a manifest-declared directory field to absolute paths within the + * plugin root. The Claude plugin manifest allows path fields to be either a + * single string or an array of strings + * (https://code.claude.com/docs/en/plugins-reference#path-behavior-rules), so + * both shapes are normalized here. + * + * The first `manifestKeys` entry that supplies at least one non-empty path wins + * (later keys are ignored — used for the `commands` > `slash-commands` legacy + * fallback). When no key is set the plugin's default subdirectory (`fallback`) + * is used. Entries that resolve outside the plugin root are dropped with a + * warning so misconfigured manifests are visible without traversal escape. + */ async function resolvePluginDir( root: ClaudePluginRoot, manifestKeys: ReadonlyArray, @@ -67,30 +80,46 @@ async function resolvePluginDir( const manifest = await readPluginManifest(root); const fallbackDir = path.join(root.path, fallback); - let configured: string | undefined; + let configured: string[] | undefined; let matchedKey: keyof ClaudePluginManifest | undefined; for (const key of manifestKeys) { const val = manifest?.[key]; - if (typeof val === "string" && val.trim()) { - configured = val.trim(); + const candidates: string[] = []; + if (typeof val === "string") { + const trimmed = val.trim(); + if (trimmed) candidates.push(trimmed); + } else if (Array.isArray(val)) { + for (const entry of val) { + if (typeof entry !== "string") continue; + const trimmed = entry.trim(); + if (trimmed) candidates.push(trimmed); + } + } + if (candidates.length > 0) { + configured = candidates; matchedKey = key; break; } } if (configured === undefined) { - return { dir: fallbackDir }; + return { dirs: [fallbackDir], warnings: [] }; } - const resolved = path.resolve(root.path, configured); - if (isWithinPluginRoot(root.path, resolved)) { - return { dir: resolved }; + const dirs: string[] = []; + const warnings: string[] = []; + for (const entry of configured) { + const resolved = path.resolve(root.path, entry); + if (isWithinPluginRoot(root.path, resolved)) { + dirs.push(resolved); + } else { + warnings.push( + `[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${entry}`, + ); + } } - return { - dir: fallbackDir, - warning: `[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${configured}`, - }; + return { dirs, warnings }; } // ============================================================================= @@ -104,24 +133,30 @@ async function loadSkills(ctx: LoadContext): Promise> { warnings.push(...rootWarnings); const results = await Promise.all( roots.map(async root => { - const { dir: skillsDir, warning } = await resolvePluginDir(root, ["skills"], "skills"); - const result = await scanSkillsFromDir(ctx, { - dir: skillsDir, - providerId: PROVIDER_ID, - level: root.scope, - }); - return { root, result, warning }; + const { dirs: skillsDirs, warnings: resolveWarnings } = await resolvePluginDir(root, ["skills"], "skills"); + const scanResults = await Promise.all( + skillsDirs.map(dir => + scanSkillsFromDir(ctx, { + dir, + providerId: PROVIDER_ID, + level: root.scope, + }), + ), + ); + return { scanResults, resolveWarnings }; }), ); - for (const { result, warning } of results) { - if (warning) warnings.push(warning); + for (const { scanResults, resolveWarnings } of results) { + warnings.push(...resolveWarnings); // Intentionally do NOT prefix skill names with `root.plugin`. // The `plugin:name` format breaks skill:// URL parsing (colons are // ambiguous with port separators) and is unintuitive for callers. // Dedup-by-key in the capability layer already handles name collisions // across providers using priority ordering. - items.push(...result.items); - if (result.warnings) warnings.push(...result.warnings); + for (const result of scanResults) { + items.push(...result.items); + if (result.warnings) warnings.push(...result.warnings); + } } return { items, warnings }; } @@ -139,28 +174,38 @@ async function loadSlashCommands(ctx: LoadContext): Promise { - const { dir: commandsDir, warning } = await resolvePluginDir(root, ["commands", "slash-commands"], "commands"); - const commandResult = await loadFilesFromDir(ctx, commandsDir, PROVIDER_ID, root.scope, { - extensions: ["md"], - transform: (name, content, filePath, source) => { - const cmdName = name.replace(/\.md$/, ""); - return { - name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName, - path: filePath, - content, - level: root.scope, - _source: source, - }; - }, - }); - return { commandResult, warning }; + const { dirs: commandsDirs, warnings: resolveWarnings } = await resolvePluginDir( + root, + ["commands", "slash-commands"], + "commands", + ); + const commandResults = await Promise.all( + commandsDirs.map(dir => + loadFilesFromDir(ctx, dir, PROVIDER_ID, root.scope, { + extensions: ["md"], + transform: (name, content, filePath, source) => { + const cmdName = name.replace(/\.md$/, ""); + return { + name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName, + path: filePath, + content, + level: root.scope, + _source: source, + }; + }, + }), + ), + ); + return { commandResults, resolveWarnings }; }), ); - for (const { commandResult, warning } of results) { - if (warning) warnings.push(warning); - items.push(...commandResult.items); - if (commandResult.warnings) warnings.push(...commandResult.warnings); + for (const { commandResults, resolveWarnings } of results) { + warnings.push(...resolveWarnings); + for (const commandResult of commandResults) { + items.push(...commandResult.items); + if (commandResult.warnings) warnings.push(...commandResult.warnings); + } } return { items, warnings }; diff --git a/packages/coding-agent/test/discovery/claude-plugins.test.ts b/packages/coding-agent/test/discovery/claude-plugins.test.ts index ba39fe344..31a0cd867 100644 --- a/packages/coding-agent/test/discovery/claude-plugins.test.ts +++ b/packages/coding-agent/test/discovery/claude-plugins.test.ts @@ -645,6 +645,132 @@ describe("listClaudePluginRoots", () => { expect(found).toBeUndefined(); }); + + test("reads slash commands from array-form commands manifest field (Claude plugin path-behavior rules)", async () => { + // Mirrors real-world plugins such as addyosmani/agent-skills whose plugin.json + // declares `"commands": ["./.claude/commands", "./commands"]`. Both directories + // contribute; each command lands under the plugin's namespace. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-array"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude", "commands"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-array@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./.claude/commands", "./commands"] }), + ); + await fs.writeFile(path.join(pluginPath, ".claude", "commands", "spec.md"), "Spec\n"); + await fs.writeFile(path.join(pluginPath, ".claude", "commands", "plan.md"), "Plan\n"); + await fs.writeFile(path.join(pluginPath, "commands", "review.md"), "Review\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + const names = result.all + .filter(command => command.name.startsWith("manifest-commands-array:")) + .map(command => command.name) + .sort(); + expect(names).toEqual([ + "manifest-commands-array:plan", + "manifest-commands-array:review", + "manifest-commands-array:spec", + ]); + }); + + test("array-form commands warns on out-of-root entries while loading valid ones", async () => { + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-mixed"); + const outsideDir = path.join(tempDir, "outside-commands"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude", "commands"), { recursive: true }); + await fs.mkdir(outsideDir, { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-mixed@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./.claude/commands", "../../outside-commands"] }), + ); + await fs.writeFile(path.join(pluginPath, ".claude", "commands", "spec.md"), "Spec\n"); + await fs.writeFile(path.join(outsideDir, "escape.md"), "Escape\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings.some(w => w.includes("Ignoring commands path outside plugin root"))).toBe(true); + expect(result.all.find(c => c.name === "manifest-commands-mixed:spec")).toBeDefined(); + expect(result.all.find(c => c.name === "manifest-commands-mixed:escape")).toBeUndefined(); + }); + + test("reads skills from array-form skills manifest field", async () => { + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-array"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "extra-skills", "alpha"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "more-skills", "beta"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-array@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./extra-skills", "./more-skills"] }), + ); + await fs.writeFile( + path.join(pluginPath, "extra-skills", "alpha", "SKILL.md"), + "---\nname: alpha\ndescription: Alpha skill\n---\nBody\n", + ); + await fs.writeFile( + path.join(pluginPath, "more-skills", "beta", "SKILL.md"), + "---\nname: beta\ndescription: Beta skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "alpha")).toBeDefined(); + expect(result.all.find(s => s.name === "beta")).toBeDefined(); + }); }); describe("discoverAgents plugin precedence", () => { From f276d80fc46f3ca2fb0ba9cbea97adbccfe01b3a Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 12:09:22 +0000 Subject: [PATCH 12/91] fix(discovery): honour per-field Claude plugin path merge semantics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review feedback on #4610: array-form skills was silently replacing the default `skills/` scan when the manifest declared any explicit entries. Per the Claude plugins reference "Path behavior rules" (https://code.claude.com/docs/en/plugins-reference#path-behavior-rules): - `skills` ADDS to the default `skills/` scan - `commands` / `slash-commands` REPLACE the default `commands/` scan `resolvePluginDir` now takes an explicit `includeFallback` flag. `loadSkills` passes `true` (fallback + declared entries, deduped by resolved absolute path so a manifest may still list `./skills` alongside extras without double-load); `loadSlashCommands` passes `false` (replace semantic preserved). Deduplication keeps the fallback first and declared entries in manifest order. Regression tests cover both semantics: skills-array merges with default `skills/`; commands-array replaces default `commands/` so a stray `commands/default.md` no longer loads once the manifest declares an alternative — matching Claude's documented behavior. --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/discovery/claude-plugins.ts | 55 +++++++++--- .../test/discovery/claude-plugins.test.ts | 86 +++++++++++++++++++ 3 files changed, 129 insertions(+), 14 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 2d7ad6da5..98b991833 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry so misconfigured paths remain visible. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609)) +- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry. The resolver also now honours Claude's per-field merge semantic — `skills` adds to the default `skills/` scan; `commands`/`slash-commands` replace the default `commands/` — so plugins like `{"skills":["./extra-skills"]}` no longer lose their default `skills/` folder while `{"commands":["./admin"]}` still replaces `commands/` as documented. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609)) - Fixed macOS Backspace on empty search not deleting sessions in the `/resume` picker; Fn+Backspace terminals that deliver `\x7f` instead of `\e[3~` now reach the delete confirmation dialog. ([#4580](https://github.com/can1357/oh-my-pi/pull/4580) by [@JagravNaik](https://github.com/JagravNaik)) - Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal session title text. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) - Fixed empty session `.jsonl` files accumulating in `~/.omp/agent/sessions//` after a draft-then-clear exit cycle. `SessionManager.saveDraft(text)` materializes the session file so the draft sidecar has a parent; a subsequent `saveDraft("")` unlinked the sidecar but left the metadata-only JSONL behind (title slot + session header + startup selector entries, ~500–750 B), and `#shouldHaveSessionFile()` could no longer prune it once `#fileIsCurrent`/`#forceFileCreation` were latched. `SessionManager.close()` now drops only draft-owned metadata-only sessions with no saved draft sidecar to reattach to, while keeping real conversations, meaningful non-message entries such as handoff custom messages, explicit `ensureOnDisk()` sessions, drafts still pending for `--resume`, and never-materialized sessions untouched ([#4571](https://github.com/can1357/oh-my-pi/issues/4571)). diff --git a/packages/coding-agent/src/discovery/claude-plugins.ts b/packages/coding-agent/src/discovery/claude-plugins.ts index 8c46b7112..f9647c124 100644 --- a/packages/coding-agent/src/discovery/claude-plugins.ts +++ b/packages/coding-agent/src/discovery/claude-plugins.ts @@ -61,21 +61,33 @@ function isWithinPluginRoot(rootPath: string, targetPath: string): boolean { /** * Resolve a manifest-declared directory field to absolute paths within the - * plugin root. The Claude plugin manifest allows path fields to be either a - * single string or an array of strings - * (https://code.claude.com/docs/en/plugins-reference#path-behavior-rules), so - * both shapes are normalized here. + * plugin root. * - * The first `manifestKeys` entry that supplies at least one non-empty path wins - * (later keys are ignored — used for the `commands` > `slash-commands` legacy - * fallback). When no key is set the plugin's default subdirectory (`fallback`) - * is used. Entries that resolve outside the plugin root are dropped with a - * warning so misconfigured manifests are visible without traversal escape. + * Manifest path fields may be `string` or `string[]` + * (https://code.claude.com/docs/en/plugins-reference#path-behavior-rules); + * both shapes are normalized here. The first `manifestKeys` entry that + * supplies at least one non-empty path wins (later keys are ignored — used for + * the `commands` > `slash-commands` legacy fallback). + * + * `fallback` is the default subdirectory (e.g. `skills/`, `commands/`) and + * `includeFallback` controls the Claude-documented merge semantic per field: + * + * - `skills` **adds to** the default: `fallback` is always scanned, and any + * manifest entries load alongside it. Callers pass `includeFallback: true`. + * - `commands` / `slash-commands` **replace** the default: an explicit + * manifest key means the default `commands/` directory is not scanned. + * Callers pass `includeFallback: false` (the manifest itself may still + * list `./commands` explicitly to keep it). + * + * When no matching key is set, the fallback is used regardless. Entries that + * resolve outside the plugin root are dropped with a warning so misconfigured + * manifests remain observable and cannot escape via traversal. */ async function resolvePluginDir( root: ClaudePluginRoot, manifestKeys: ReadonlyArray, fallback: string, + includeFallback: boolean, ): Promise { const manifest = await readPluginManifest(root); const fallbackDir = path.join(root.path, fallback); @@ -106,17 +118,28 @@ async function resolvePluginDir( return { dirs: [fallbackDir], warnings: [] }; } + // Dedup preserves order: default entry (when included) first, then declared + // entries in manifest order. Deduping the paths themselves means a plugin + // author can still list `./commands` explicitly when they want the default + // alongside extras without producing double-loads. + const seen = new Set(); const dirs: string[] = []; const warnings: string[] = []; + if (includeFallback) { + seen.add(fallbackDir); + dirs.push(fallbackDir); + } for (const entry of configured) { const resolved = path.resolve(root.path, entry); - if (isWithinPluginRoot(root.path, resolved)) { - dirs.push(resolved); - } else { + if (!isWithinPluginRoot(root.path, resolved)) { warnings.push( `[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${entry}`, ); + continue; } + if (seen.has(resolved)) continue; + seen.add(resolved); + dirs.push(resolved); } return { dirs, warnings }; @@ -133,7 +156,12 @@ async function loadSkills(ctx: LoadContext): Promise> { warnings.push(...rootWarnings); const results = await Promise.all( roots.map(async root => { - const { dirs: skillsDirs, warnings: resolveWarnings } = await resolvePluginDir(root, ["skills"], "skills"); + const { dirs: skillsDirs, warnings: resolveWarnings } = await resolvePluginDir( + root, + ["skills"], + "skills", + true, + ); const scanResults = await Promise.all( skillsDirs.map(dir => scanSkillsFromDir(ctx, { @@ -178,6 +206,7 @@ async function loadSlashCommands(ctx: LoadContext): Promise diff --git a/packages/coding-agent/test/discovery/claude-plugins.test.ts b/packages/coding-agent/test/discovery/claude-plugins.test.ts index 31a0cd867..f2a4c5d85 100644 --- a/packages/coding-agent/test/discovery/claude-plugins.test.ts +++ b/packages/coding-agent/test/discovery/claude-plugins.test.ts @@ -771,6 +771,92 @@ describe("listClaudePluginRoots", () => { expect(result.all.find(s => s.name === "alpha")).toBeDefined(); expect(result.all.find(s => s.name === "beta")).toBeDefined(); }); + + test("manifest skills field merges with default skills/ directory (adds, not replaces)", async () => { + // Per Claude plugins reference "Path behavior rules": + // `skills` adds to the default `skills/` scan; the default is always loaded + // alongside any manifest-declared directories. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-merge"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "skills", "default-skill"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "extra-skills", "extra-skill"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-merge@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./extra-skills"] }), + ); + await fs.writeFile( + path.join(pluginPath, "skills", "default-skill", "SKILL.md"), + "---\nname: default-skill\ndescription: Default skill\n---\nBody\n", + ); + await fs.writeFile( + path.join(pluginPath, "extra-skills", "extra-skill", "SKILL.md"), + "---\nname: extra-skill\ndescription: Extra skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "default-skill")).toBeDefined(); + expect(result.all.find(s => s.name === "extra-skill")).toBeDefined(); + }); + + test("manifest commands field replaces default commands/ directory (Claude replace semantics)", async () => { + // Per Claude plugins reference "Path behavior rules": + // `commands` REPLACES the default `commands/` scan when the manifest key is set. + // A plugin that wants both must list `./commands` explicitly. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-replace"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "admin-commands"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-replace@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./admin-commands"] }), + ); + // This file lives under the default commands/ dir and MUST NOT load once the + // manifest declares `commands` (Claude's documented "replaces default" semantic). + await fs.writeFile(path.join(pluginPath, "commands", "default.md"), "Default\n"); + await fs.writeFile(path.join(pluginPath, "admin-commands", "admin.md"), "Admin\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(c => c.name === "manifest-commands-replace:admin")).toBeDefined(); + expect(result.all.find(c => c.name === "manifest-commands-replace:default")).toBeUndefined(); + }); }); describe("discoverAgents plugin precedence", () => { From c42d2b290741e6c2abb637d44613b4b7cf1c1085 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 12:16:02 +0000 Subject: [PATCH 13/91] fix(discovery): load file-form Claude plugin commands Claude plugin manifests allow command path entries to name either directories or flat `.md` command files. The array resolver was now preserving those file paths, but `loadSlashCommands` still sent every resolved entry through `loadFilesFromDir`, which only globs inside directories. A manifest such as `{"commands":["./custom/deploy.md"]}` therefore replaced the default scan and then loaded nothing. Teach the command loader to stat each resolved entry: `.md` files are read as single slash commands with the same plugin namespace and source metadata as directory-loaded files; directories continue through `loadFilesFromDir`. Missing entries keep the existing silent-empty behavior. Add a regression test covering a mixed array of a direct command file and a command directory while proving default `commands/` remains replaced unless listed explicitly. --- .../src/discovery/claude-plugins.ts | 32 ++++++++++++-- .../test/discovery/claude-plugins.test.ts | 42 +++++++++++++++++++ 2 files changed, 70 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/discovery/claude-plugins.ts b/packages/coding-agent/src/discovery/claude-plugins.ts index f9647c124..24e938ce7 100644 --- a/packages/coding-agent/src/discovery/claude-plugins.ts +++ b/packages/coding-agent/src/discovery/claude-plugins.ts @@ -4,6 +4,7 @@ * Loads configuration from ~/.claude/plugins/cache/ based on installed_plugins.json registry. * Priority: 70 (below claude.ts at 80, so user overrides in .claude/ take precedence) */ +import * as fs from "node:fs/promises"; import * as path from "node:path"; import { logger } from "@oh-my-pi/pi-utils"; import { registerProvider } from "../capability"; @@ -209,8 +210,31 @@ async function loadSlashCommands(ctx: LoadContext): Promise - loadFilesFromDir(ctx, dir, PROVIDER_ID, root.scope, { + commandsDirs.map(async dir => { + try { + const stats = await fs.stat(dir); + if (stats.isFile()) { + if (path.extname(dir) !== ".md") return { items: [], warnings: [] }; + const content = await readFile(dir); + if (content === null) return { items: [], warnings: [`Failed to read file: ${dir}`] }; + const cmdName = path.basename(dir).replace(/\.md$/, ""); + return { + items: [ + { + name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName, + path: dir, + content, + level: root.scope, + _source: createSourceMeta(PROVIDER_ID, dir, root.scope), + }, + ], + warnings: [], + }; + } + } catch { + // Missing entries behave like missing directories: no items, no warning. + } + return loadFilesFromDir(ctx, dir, PROVIDER_ID, root.scope, { extensions: ["md"], transform: (name, content, filePath, source) => { const cmdName = name.replace(/\.md$/, ""); @@ -222,8 +246,8 @@ async function loadSlashCommands(ctx: LoadContext): Promise { ]); }); + test("reads slash commands from array-form manifest file entries", async () => { + // Claude plugins reference allows command paths to be either flat `.md` + // files or directories. A manifest-declared commands field still replaces + // default `commands/`; plugins that want defaults must list `./commands`. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-files"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "custom"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "ops"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-files@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./custom/deploy.md", "./ops"] }), + ); + await fs.writeFile(path.join(pluginPath, "custom", "deploy.md"), "Deploy\n"); + await fs.writeFile(path.join(pluginPath, "ops", "rollback.md"), "Rollback\n"); + await fs.writeFile(path.join(pluginPath, "commands", "default.md"), "Default\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(c => c.name === "manifest-commands-files:deploy")?.content).toBe("Deploy\n"); + expect(result.all.find(c => c.name === "manifest-commands-files:rollback")?.content).toBe("Rollback\n"); + expect(result.all.find(c => c.name === "manifest-commands-files:default")).toBeUndefined(); + }); + test("array-form commands warns on out-of-root entries while loading valid ones", async () => { const pluginsDir = path.join(tempDir, ".claude", "plugins"); const pluginPath = path.join(tempDir, "plugins", "manifest-commands-mixed"); From c92cf0a9054e2384cb3224ec49c79535a35d9f47 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 12:23:18 +0000 Subject: [PATCH 14/91] fix(discovery): load Claude plugin skill paths that point at a SKILL.md directory MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Claude plugin manifests may declare a `skills` path that resolves directly to a directory whose `SKILL.md` IS the skill (e.g. `"skills": ["./"]` or a subdirectory containing only `SKILL.md`). `scanSkillsFromDir` only scanned `//SKILL.md` children, so the single-skill directory layout — the common shape the Claude plugins reference documents for plugins shipping one skill — silently dropped every array-form manifest entry that pointed at it. Add an opt-in `includeSelf` flag to `ScanSkillsFromDirOptions`: when set, `/SKILL.md` (if present) is loaded as a skill in addition to the existing child scan. The Claude plugin skills loader opts in; every other provider (agents, builtin, claude.ts, codex, github, omp-plugins, opencode) keeps the strict child-scan semantic they rely on. Frontmatter `name` still wins over the directory basename fallback. Regression test: `skills: ["./single"]` where `./single/SKILL.md` is the only skill file loads the skill under its frontmatter name. --- .../src/discovery/claude-plugins.ts | 1 + .../coding-agent/src/discovery/helpers.ts | 17 +++++++- .../test/discovery/claude-plugins.test.ts | 39 +++++++++++++++++++ 3 files changed, 56 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/discovery/claude-plugins.ts b/packages/coding-agent/src/discovery/claude-plugins.ts index 24e938ce7..7cfd8ad83 100644 --- a/packages/coding-agent/src/discovery/claude-plugins.ts +++ b/packages/coding-agent/src/discovery/claude-plugins.ts @@ -169,6 +169,7 @@ async function loadSkills(ctx: LoadContext): Promise> { dir, providerId: PROVIDER_ID, level: root.scope, + includeSelf: true, }), ), ); diff --git a/packages/coding-agent/src/discovery/helpers.ts b/packages/coding-agent/src/discovery/helpers.ts index d1e6e7c91..8535fa0ad 100644 --- a/packages/coding-agent/src/discovery/helpers.ts +++ b/packages/coding-agent/src/discovery/helpers.ts @@ -312,6 +312,15 @@ export interface ScanSkillsFromDirOptions { providerId: string; level: "user" | "project"; requireDescription?: boolean; + /** + * When true, treat a `SKILL.md` sitting directly under `dir` as a single skill in addition to + * scanning `//SKILL.md` children. Matches the Claude plugin manifest convention + * that lets a skill path point at a directory containing `SKILL.md` directly (e.g. + * `"skills": ["./"]`), where the frontmatter `name` determines the invocation name and the + * directory basename is the fallback. Default `false` preserves the strict child-scan + * semantic every non-Claude provider relies on. + */ + includeSelf?: boolean; } // Stable ordering used for skill lists in prompts: name (case-insensitive), then name, then path. @@ -368,7 +377,13 @@ export async function scanSkillsFromDir( } }; - const work = []; + const work: Promise[] = []; + if (options.includeSelf) { + const selfSkillPath = path.join(dir, "SKILL.md"); + if (fs.existsSync(selfSkillPath)) { + work.push(loadSkill(selfSkillPath)); + } + } for (const entry of entries) { if (entry.name.startsWith(".")) continue; if (!entry.isDirectory() && !entry.isSymbolicLink()) continue; diff --git a/packages/coding-agent/test/discovery/claude-plugins.test.ts b/packages/coding-agent/test/discovery/claude-plugins.test.ts index a48ca31d3..b7afdd239 100644 --- a/packages/coding-agent/test/discovery/claude-plugins.test.ts +++ b/packages/coding-agent/test/discovery/claude-plugins.test.ts @@ -859,6 +859,45 @@ describe("listClaudePluginRoots", () => { expect(result.all.find(s => s.name === "extra-skill")).toBeDefined(); }); + test("array-form skills entry pointing at a directory containing SKILL.md loads the single skill", async () => { + // Per Claude plugins reference: a skills path may point directly at a directory whose + // SKILL.md is the skill (frontmatter name → invocation, directory basename → fallback). + // Real plugins use `"skills": ["./"]` — that entry must not silently drop the skill. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-self"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "single"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-self@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./single"] }), + ); + await fs.writeFile( + path.join(pluginPath, "single", "SKILL.md"), + "---\nname: solo-skill\ndescription: Solo skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "solo-skill")).toBeDefined(); + }); + test("manifest commands field replaces default commands/ directory (Claude replace semantics)", async () => { // Per Claude plugins reference "Path behavior rules": // `commands` REPLACES the default `commands/` scan when the manifest key is set. From 1e937cc7567a6460bf242b86a2b114163bb2de14 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 13:04:15 +0000 Subject: [PATCH 15/91] fix(coding-agent): refreshed advisor on role model changes Rebuilt active advisor runtimes when modelRoles.advisor changes so live sessions stop using stale advisor models. Added regression coverage for advisor role updates reaching the live advisor without a manual /advisor restart. Fixes #4612 --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/config/settings.ts | 9 +++++++++ .../coding-agent/src/session/agent-session.ts | 17 ++++++++++++++++- .../coding-agent/test/advisor-toggle.test.ts | 12 ++++++++++++ 4 files changed, 38 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c03541a2b..bd1c5f2af 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -23,6 +23,7 @@ - Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal text. - Fixed empty session `.jsonl` files accumulating in the sessions directory after a draft-then-clear exit cycle. - Fixed advisor being disabled for the entire session when resolving to a reasoning model with no controllable effort surface (e.g., `devin/glm-5-2*`). +- Fixed live advisors continuing to use a stale `modelRoles.advisor` selection after `/model` changed the advisor model. ([#4612](https://github.com/can1357/oh-my-pi/issues/4612)) - Fixed legacy extension plugin validation failures by re-exporting relocated catalog symbols (such as `calculateCost`, `modelsAreEqual`, and `getBundledProviders`) through the legacy `pi-ai` root shim. - Fixed legacy Pi extension imports of `DefaultResourceLoader` by adding a compatibility loader shim that translates `resourceLoader` into native session discovery options. - Fixed legacy Pi extension reloads on POSIX to ensure same-process re-imports pick up edits across the entire dependency graph. diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index 0bde19651..268b147f7 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -429,6 +429,9 @@ export class Settings { if (path === "statusLine.sessionAccent") { statusLineSessionAccentSignal.fire(); } + if (path === "modelRoles") { + modelRolesSignal.fire(); + } } /** @@ -1477,6 +1480,12 @@ const appendOnlyModeSignal = new SettingSignal<[value: string]>("provider.append */ export const onAppendOnlyModeChanged = (cb: (value: string) => void) => appendOnlyModeSignal.on(cb); +/** Fires when any model role changes at runtime. */ +const modelRolesSignal = new SettingSignal("modelRoles"); + +/** Subscribe to model role changes. Returns an unsubscribe function. */ +export const onModelRolesChanged: (cb: () => void) => () => void = modelRolesSignal.on.bind(modelRolesSignal); + /** Fires when `statusLine.sessionAccent` changes at runtime. */ const statusLineSessionAccentSignal = new SettingSignal("statusLine.sessionAccent"); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index dad37f8bf..685d7782e 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -179,7 +179,12 @@ import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles"; import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates"; import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; import type { Settings, SkillsSettings } from "../config/settings"; -import { getDefault, onAppendOnlyModeChanged, validateProviderMaxInFlightRequests } from "../config/settings"; +import { + getDefault, + onAppendOnlyModeChanged, + onModelRolesChanged, + validateProviderMaxInFlightRequests, +} from "../config/settings"; import { RawSseDebugBuffer } from "../debug/raw-sse-buffer"; import { loadCapability } from "../discovery"; import { expandApplyPatchToEntries, normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit"; @@ -1552,6 +1557,7 @@ export class AgentSession { #cancelExitRecorder?: () => void; #exitRecorded = false; #unsubscribeAppendOnly?: () => void; + #unsubscribeModelRoles?: () => void; /** Last (enable, providerId) tuple resolved by `#syncAppendOnlyContext` — used to skip no-op invalidations. */ #lastAppendOnlyResolution?: { enable: boolean; providerId: string | undefined }; #eventListeners: AgentSessionEventListener[] = []; @@ -2249,6 +2255,11 @@ export class AgentSession { this.#unsubscribeAgent = this.agent.subscribe(this.#handleAgentEvent); // Re-evaluate append-only context mode when the setting changes at runtime. this.#unsubscribeAppendOnly = onAppendOnlyModeChanged(_value => this.#syncAppendOnlyContext(this.model)); + this.#unsubscribeModelRoles = onModelRolesChanged(() => { + if (!this.#advisorEnabled || this.#isDisposed) return; + if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime(); + this.#buildAdvisorRuntime(true); + }); } // ------------------------------------------------------------------------- // Advisor runtime lifecycle @@ -5735,6 +5746,10 @@ export class AgentSession { this.#unsubscribeAppendOnly(); this.#unsubscribeAppendOnly = undefined; } + if (this.#unsubscribeModelRoles) { + this.#unsubscribeModelRoles(); + this.#unsubscribeModelRoles = undefined; + } this.#eventListeners = []; } diff --git a/packages/coding-agent/test/advisor-toggle.test.ts b/packages/coding-agent/test/advisor-toggle.test.ts index 0812a895b..0d911aad0 100644 --- a/packages/coding-agent/test/advisor-toggle.test.ts +++ b/packages/coding-agent/test/advisor-toggle.test.ts @@ -98,6 +98,18 @@ describe("AgentSession advisor toggle", () => { expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); }); + it("refreshes the live advisor when the advisor role setting changes", () => { + session.settings.setModelRole("advisor", `${model.provider}/${model.id}`); + expect(session.setAdvisorEnabled(true)).toBe(true); + expect(session.getAdvisorAgent()?.state.model.provider).toBe(model.provider); + expect(session.getAdvisorAgent()?.state.model.id).toBe(model.id); + + session.settings.setModelRole("advisor", `${replacementModel.provider}/${replacementModel.id}`); + + expect(session.getAdvisorAgent()?.state.model.provider).toBe(replacementModel.provider); + expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); + }); + it("keeps explicit enable idempotent when the advisor config is unchanged", () => { session.settings.setModelRole("advisor", `${model.provider}/${model.id}`); expect(session.setAdvisorEnabled(true)).toBe(true); From 44daed1fffdacab35576601aa2cb1e4c1ae9bfc1 Mon Sep 17 00:00:00 2001 From: chan1103 Date: Sun, 5 Jul 2026 23:23:12 +0900 Subject: [PATCH 16/91] fix(coding-agent/edit): sealed inverse video and preserved gutters in wrapped diff rows MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two wrap artifacts in the Edit result card, both in wrapEditRendererLine: - A row that broke inside an intra-line diff highlight ended with inverse video still active (only the foreground was reset), so the frame's right-edge padding painted as a default-foreground block. Every wrapped diff row now closes inverse alongside the foreground reset; the next row re-opens its own state, so highlights spanning the break render the same. - The gutter matcher required a marker at column 0 immediately followed by digits, a shape only produced when marker and number exactly fill the gutter. Left-padded gutters (" -42│", any line number narrower than the widest in the diff) and dedup-blanked gutters (" +│" on the added row of a single-line replacement) fell back to generic wrapping, so their continuation rows escaped into the line-number column. │-separated gutters now accept padded and blank line numbers; ASCII "|" gutters still require the canonical marker+number shape emitted by the plain fallback, so body lines that merely start with "|", " |", or "123|" keep wrapping generically. Regression tests cover continuation-gutter containment, net-inverse-off at every row end (with a precondition proving a highlight actually crossed a break), phantom-gutter rejection for pipe- and digit-leading body lines, and the plain-fallback canonical-row path. --- packages/coding-agent/CHANGELOG.md | 5 + packages/coding-agent/src/edit/renderer.ts | 24 ++- .../test/tools/edit-renderer.test.ts | 171 ++++++++++++++++++ 3 files changed, 195 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 46b71b427..bfa1dee60 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Fixed + +- Fixed wrapped Edit-diff rows leaking inverse video into the result card's right-edge padding: a row that broke inside an intra-line highlight left inverse active at the row end, so the frame padding after it rendered as a default-foreground block ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) +- Fixed Edit-diff continuation rows escaping into the line-number column when the row's gutter was left-padded (line number narrower than the widest in the diff) or blanked by the gutter dedup (the bare `+` row of a single-line replacement); such rows now wrap behind a continuation gutter, while body lines that merely start with `|` keep wrapping generically ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) + ## [16.3.7] - 2026-07-05 ### Added diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index e02af1eb7..14682a59e 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -679,21 +679,35 @@ function wrapEditRendererLine(line: string, width: number): string[] { const startAnsi = line.match(/^((?:\x1b\[[0-9;]*m)*)/)?.[1] ?? ""; const bodyWithReset = line.slice(startAnsi.length); const body = bodyWithReset.endsWith("\x1b[39m") ? bodyWithReset.slice(0, -"\x1b[39m".length) : bodyWithReset; - const diffMatch = /^([+\-\s])(\s*\d+)([|│])(.*)$/s.exec(body); + // Gutter shapes produced by formatCodeFrameLine: "-315│", " 313│", "+322│", + // plus the deduplicated forms " +│" and " │" whose repeated line number + // renderDiff blanked (single-line replacement pairs and insert-then-context + // runs) — all │-separated. ASCII "|" gutters exist only in raw canonical + // diff rows passed through by the plain fallback ("-42|old", " 42|ctx"), + // which always carry a marker column ("+"/"-"/space) and a line number. So + // the number is optional for "│", while "|" requires the full canonical + // shape; anything else (a body line merely starting with "|", error text + // like "123|…") is not a diff row and wraps generically. + const diffMatch = /^(\s*[+-]?\s*\d*)([|│])(.*)$/s.exec(body); - if (!diffMatch) { + if (!diffMatch || diffMatch[1].length === 0 || (diffMatch[2] === "|" && !/^[+\-\s]\s*\d+$/.test(diffMatch[1]))) { return wrapTextWithAnsi(line, width); } - const [, marker, lineNum, separator, content] = diffMatch; - const prefix = `${marker}${lineNum}${separator}`; + const [, gutter, separator, content] = diffMatch; + const prefix = `${gutter}${separator}`; const prefixWidth = visibleWidth(prefix); const contentWidth = Math.max(1, width - prefixWidth); const continuationPrefix = `${" ".repeat(Math.max(0, prefixWidth - 1))}${separator}`; const wrappedContent = wrapTextWithAnsi(content ?? "", contentWidth); + // Each visual row is a standalone terminal line: wrapTextWithAnsi re-opens + // active SGR state at the next row's start, so a row that breaks inside an + // intra-line diff highlight still ends with inverse video active. Close it + // alongside the foreground reset — otherwise the frame padding appended + // after the row is painted as an inverse block (default-foreground cells). return wrappedContent.map( - (segment, index) => `${startAnsi}${index === 0 ? prefix : continuationPrefix}${segment}\x1b[39m`, + (segment, index) => `${startAnsi}${index === 0 ? prefix : continuationPrefix}${segment}\x1b[27m\x1b[39m`, ); } diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index 7962c7353..7b84723cf 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -7,10 +7,12 @@ import type { AgentTool } from "@oh-my-pi/pi-agent-core"; import { renderGalleryState, resolveFixture } from "@oh-my-pi/pi-coding-agent/cli/gallery-cli"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { editToolRenderer } from "@oh-my-pi/pi-coding-agent/edit/renderer"; +import { renderDiff } from "@oh-my-pi/pi-coding-agent/modes/components/diff"; import { ToolExecutionComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tool-execution"; import * as themeModule from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { Text, type TUI, visibleWidth } from "@oh-my-pi/pi-tui"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import chalk from "chalk"; beforeAll(async () => { resetSettingsForTest(); @@ -519,3 +521,172 @@ describe("editToolRenderer", () => { expect(text).not.toContain("No changes"); }); }); + +describe("editToolRenderer diff line wrapping", () => { + // Renders a completed single-line replacement (`-N|old` + `+N|new`) through + // the real renderDiff so the result carries its production shapes: a blanked + // dedup gutter on the `+` row (` +│`) and intra-line inverse highlights. + async function renderSingleLineReplacement( + oldLine: string, + newLine: string, + width: number, + ): Promise { + const uiTheme = await getUiTheme(); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: "Updated demo.ts" }], + details: { diff: `-42|${oldLine}\n+42|${newLine}`, op: "update", path: "demo.ts" }, + }, + { expanded: true, isPartial: false, renderContext: { renderDiff } }, + uiTheme, + { file_path: "demo.ts" }, + ); + return component.render(width); + } + + /** Net SGR inverse state after scanning a row; 38/48 extended-color args must not be misread as attribute 7. */ + function inverseActiveAtRowEnd(row: string): boolean { + let inverse = false; + for (const match of row.matchAll(/\x1b\[([0-9;]*)m/g)) { + const params = match[1].split(";"); + for (let i = 0; i < params.length; i++) { + const param = params[i]; + if (param === "38" || param === "48") { + i += params[i + 1] === "2" ? 4 : params[i + 1] === "5" ? 2 : 0; + } else if (param === "" || param === "0") inverse = false; + else if (param === "7") inverse = true; + else if (param === "27") inverse = false; + } + } + return inverse; + } + + it("keeps added-line continuation rows inside the blanked dedup gutter", async () => { + // renderDiff blanks the repeated line number on the `+` row of a + // single-line replacement (` +│`); the wrapper must still recognize that + // gutter instead of falling back to generic wrapping at column 0. + const rows = ( + await renderSingleLineReplacement( + " the previous synopsis paragraph rambled across quarterly reconciliation notes enumerating every provisional ledger amendment the archival committee had deferred pending review by the regional custodians during the extended winter recess of the auditing season", + " the revised synopsis paragraph now catalogues seasonal festival logistics enumerating lantern shipments drum rehearsals and ribbon inventories that the parade stewards confirmed before dawn, closing with the zephyrQuota tally and the marbledFinale banner", + 100, + ) + ).map(row => Bun.stripANSI(row)); + + // The tail of the added line lands on continuation rows, which must carry + // the spaces-only continuation gutter rather than start as bare prose. + const tailRows = rows.filter(row => row.includes("zephyrQuota") || row.includes("marbledFinale")); + expect(tailRows.length).toBeGreaterThanOrEqual(1); + for (const row of tailRows) expect(row).toMatch(/^│\s+│/); + // Every body row stays inside a code-frame gutter (`-42│`, ` +│`, ` │`). + for (const row of rows.slice(1, -1)) expect(row).toMatch(/^│\s*[+-]?\s*\d*│/); + }); + + it("closes inverse video at every wrapped row end so frame padding stays uninverted", async () => { + // A long contiguous rewritten phrase forces the wrap boundary to land + // inside an inverse-highlighted span; the frame pads each row with spaces, + // so any inverse still active at row end paints those cells as gray blocks. + const previousLevel = chalk.level; + chalk.level = 3; + let rows: readonly string[]; + try { + rows = await renderSingleLineReplacement( + " stanza recounts venerable chronicle passages spanning bygone dynasties whose archivists engraved ledgers onto vellum scrolls", + " stanza celebrates luminous festival processions winding through lantern boulevards while drummers herald jubilant choruses beneath cascading ribbons and fireworks", + 100, + ); + } finally { + chalk.level = previousLevel; + } + + // Precondition: some continuation row's content reopens with inverse right + // after its gutter, proving a highlighted span crossed a wrap boundary. If + // diffWords tokenization ever changes so no span crosses, this fails loudly + // instead of letting the row-end assertions pass vacuously. + expect(rows.some(row => /│\x1b\[7m/.test(row))).toBe(true); + for (const row of rows) expect(inverseActiveAtRowEnd(row)).toBe(false); + }); + + // Error results reuse the same body-line wrapper as diff rows; these tests + // pin the boundary between prose that merely looks pipe-ish and real gutters. + async function renderErrorResultRows(errorText: string): Promise { + const uiTheme = await getUiTheme(); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: errorText }], + details: { diff: "", op: "update", path: "demo.ts" }, + isError: true, + }, + { expanded: true, isPartial: false, renderContext: { renderDiff } }, + uiTheme, + { file_path: "demo.ts" }, + ); + return component.render(100).map(row => Bun.stripANSI(row)); + } + + it("does not give pipe-leading error text a phantom diff gutter when wrapping", async () => { + // Error text is not a diff row even when it starts with `|`: an empty + // gutter must wrap generically, not spawn `|` continuation prefixes. + const rows = await renderErrorResultRows( + "| pipe-leading diagnostic output that is quite long and should certainly wrap at the render width because it keeps going on and on with more words than fit in one row of the frame", + ); + const bodyRows = rows.slice(1, -1); + // Precondition: the text actually wrapped, and the `|` lead survived on row one. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│\| /); + for (const row of bodyRows.slice(1)) expect(row).not.toMatch(/^│\s*\|/); + }); + + it("wraps spaces-then-bare-pipe error text generically instead of minting a gutter", async () => { + // A digit-less ASCII "|" gutter never comes out of formatCodeFrameLine or + // canonical diff rows; indented bare-pipe error text must wrap generically. + const rows = await renderErrorResultRows( + " | indented bare-pipe diagnostic output that is quite long and should certainly wrap at the render width because it keeps going on and on with more words than fit in one row of the frame", + ); + const bodyRows = rows.slice(1, -1); + // Precondition: the text actually wrapped, and the pipe lead survived on row one. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│\s+\| /); + for (const row of bodyRows.slice(1)) expect(row).not.toMatch(/^│\s*\|/); + }); + + it("wraps digit-leading pipe error text generically when the marker column is missing", async () => { + // Canonical ASCII-pipe rows always carry a marker column (`-42|`, ` 42|`); + // `123|` prose has a digit there instead, so it is not a diff row. + const rows = await renderErrorResultRows( + "123| numbered pipe-leading diagnostic output that is quite long and should certainly wrap at the render width because it keeps going on and on with more words than fit in one row of the frame", + ); + const bodyRows = rows.slice(1, -1); + // Precondition: the text actually wrapped, and the numbered lead survived on row one. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│123\| /); + for (const row of bodyRows.slice(1)) expect(row).not.toMatch(/^│\s*\|/); + }); + + it("keeps the numbered ASCII-pipe gutter for canonical rows through the plain fallback", async () => { + // Without renderContext the plain fallback passes canonical rows through + // verbatim; a numbered "-42|" row must still take the gutter path and + // carry an " |" continuation gutter, not generic prose wrapping. + const uiTheme = await getUiTheme(); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: "Updated demo.ts" }], + details: { + diff: "-42| the previous synopsis paragraph rambled across quarterly reconciliation notes enumerating every provisional ledger amendment the archival committee had deferred pending review by the regional custodians", + op: "update", + path: "demo.ts", + }, + }, + { expanded: true, isPartial: false }, + uiTheme, + { file_path: "demo.ts" }, + ); + + const rows = component.render(100).map(row => Bun.stripANSI(row)); + const bodyRows = rows.slice(1, -1); + // Precondition: the row actually wrapped past its first visual line. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│-42\|/); + for (const row of bodyRows.slice(1)) expect(row).toMatch(/^│\s+\|/); + }); +}); From 5cdc525d0a22b75b6ba4e311360fd01555da108b Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 17:21:11 +0000 Subject: [PATCH 17/91] fix(tui): accepted mid-prompt slash skill completions - Matched trailing slash skill prefixes during autocomplete staleness checks so Tab and Enter apply the highlighted skill token. - Cancelled mid-prompt skill autocomplete immediately when Backspace removes the triggering slash. - Added regression coverage for Tab, Enter, and Backspace behavior. Fixes #4619 --- packages/tui/CHANGELOG.md | 4 ++ packages/tui/src/components/editor.ts | 41 ++++++++++---- .../test/editor-autocomplete-actions.test.ts | 54 +++++++++++++++++++ 3 files changed, 90 insertions(+), 9 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index b99d781d7..3aceebfe4 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed mid-prompt skill autocomplete so Tab and Enter accept the highlighted `/skill:` suggestion and Backspace dismisses the popup immediately after removing the triggering slash ([#4619](https://github.com/can1357/oh-my-pi/issues/4619)). + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 2eb353fd3..a1a354351 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -21,7 +21,7 @@ import { truncateToWidth, visibleWidth, } from "../utils"; -import { SelectList, type SelectListLayoutOptions, type SelectListTheme } from "./select-list"; +import { type SelectItem, SelectList, type SelectListLayoutOptions, type SelectListTheme } from "./select-list"; const AUTOCOMPLETE_SELECT_LIST_LAYOUT: SelectListLayoutOptions = { overflowSearch: false, @@ -1119,16 +1119,16 @@ export class Editor implements Component, Focusable { // If Tab was pressed, always apply the selection if (kb.matches(data, "tui.input.tab")) { + const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to buffer edits since last refresh // (destructive keys or paste can outrun the debounced update). const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); - if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor)) { + if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor, selected)) { // Autocomplete is stale - silently cancel; Tab has no fallback action here. this.#cancelAutocomplete(); return; } - const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const shouldChainSlashCommandAutocomplete = this.#isSlashCommandNameAutocompleteSelection(); const result = this.#autocompleteProvider.applyCompletion( @@ -1164,14 +1164,14 @@ export class Editor implements Component, Focusable { (kb.matches(data, "tui.input.submit") || data === "\n") && findLeadingSlashCommandStart(this.#autocompletePrefix) !== null ) { + const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to debounce const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); - if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor)) { + if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor, selected)) { // Autocomplete is stale - cancel and fall through to normal submission this.#cancelAutocomplete(); } else { - const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const result = this.#autocompleteProvider.applyCompletion( this.#state.lines, @@ -1192,14 +1192,14 @@ export class Editor implements Component, Focusable { } // If Enter was pressed on a file path, apply completion else if (kb.matches(data, "tui.input.submit") || data === "\n") { + const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to buffer edits since last refresh. const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); - if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor)) { + if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor, selected)) { // Autocomplete is stale - cancel and fall through to normal submission this.#cancelAutocomplete(); } else { - const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const result = this.#autocompleteProvider.applyCompletion( this.#state.lines, @@ -2068,8 +2068,15 @@ export class Editor implements Component, Focusable { this.#resetKillSequence(); this.#recordUndoState(); + let removedMidPromptSlashTrigger = false; + if (this.#state.cursorCol > 0) { const line = this.#state.lines[this.#state.cursorLine] || ""; + const textBeforeCursor = line.slice(0, this.#state.cursorCol); + const trailingSlashStart = findTrailingSlashCommandStart(textBeforeCursor); + removedMidPromptSlashTrigger = + trailingSlashStart === this.#state.cursorCol - 1 && + (!this.#hasOnlyWhitespaceBeforeCursorLine() || textBeforeCursor.slice(0, trailingSlashStart).trim() !== ""); // An atomic placeholder token (image/paste marker) deletes as a unit, so a single // backspace never leaves a half-eaten `[Paste #1, +30 lines` behind as stray text. const token = this.#atomicTokenAt(line, this.#state.cursorCol - 1); @@ -2109,7 +2116,12 @@ export class Editor implements Component, Focusable { // Update or re-trigger autocomplete after backspace if (this.#autocompleteState) { - this.#debouncedUpdateAutocomplete(); + if (removedMidPromptSlashTrigger) { + this.#cancelAutocomplete(); + this.onAutocompleteUpdate?.(); + } else { + this.#debouncedUpdateAutocomplete(); + } } else { // If autocomplete was cancelled (no matches), re-trigger if we're in a completable context const currentLine = this.#state.lines[this.#state.cursorLine] || ""; @@ -2876,13 +2888,24 @@ export class Editor implements Component, Focusable { * - Slash branch re-anchors when both the prefix and the current text carry a * leading slash command and the current slash token is clean (no whitespace or * inner slash), matching `applyCompletion`'s slash-branch guard. + * - Mid-prompt skill branch re-anchors when the popup item is a skill and the + * current text still ends in a trailing slash token, matching the provider's + * mid-prompt replacement branch. * - `@`-file branch re-anchors via `#extractAtPrefix`; safe when the current text * still ends in a whitespace-anchored `@`. * - Everything else is stale — accepting it would corrupt the buffer (issue #4295). */ - #autocompletePrefixMatchesCursorText(currentTextBeforeCursor: string): boolean { + #autocompletePrefixMatchesCursorText(currentTextBeforeCursor: string, item?: SelectItem | null): boolean { if (currentTextBeforeCursor === this.#autocompletePrefix) return true; + if (item?.value.startsWith("skill:") && findTrailingSlashCommandStart(this.#autocompletePrefix) !== null) { + const currentTrailingStart = findTrailingSlashCommandStart(currentTextBeforeCursor); + if (currentTrailingStart !== null) { + const token = currentTextBeforeCursor.slice(currentTrailingStart); + if (!token.includes(" ") && !token.slice(1).includes("/")) return true; + } + } + if (findLeadingSlashCommandStart(this.#autocompletePrefix) !== null) { const currentLeadingStart = findLeadingSlashCommandStart(currentTextBeforeCursor); if (currentLeadingStart !== null) { diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index b7bbca391..2b926c410 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -134,6 +134,60 @@ class SyncSlashProvider implements AutocompleteProvider { } describe("Editor Enter handler sync slash completion", () => { + const skillCommands = [ + { name: "skill:security-scan", description: "Security scan" }, + { name: "model", description: "Switch model" }, + ]; + + function createSkillEditor(): Editor { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider(new CombinedAutocompleteProvider(skillCommands, "/tmp")); + return editor; + } + + async function openMidPromptSkillAutocomplete(editor: Editor, prose: string): Promise { + editor.handleInput(prose); + editor.handleInput("/"); + await Promise.resolve(); + + expect(editor.getText()).toBe(`${prose}/`); + expect(editor.isShowingAutocomplete()).toBe(true); + } + + it("accepts a bare mid-prompt skill slash with Tab without replacing prose", async () => { + const editor = createSkillEditor(); + + await openMidPromptSkillAutocomplete(editor, "run a "); + editor.handleInput("\t"); + + expect(editor.getText()).toBe("run a /skill:security-scan "); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + + it("accepts a bare mid-prompt skill slash with Enter and submits the completed prompt", async () => { + const editor = createSkillEditor(); + let submitted = ""; + editor.onSubmit = text => { + submitted = text; + }; + + await openMidPromptSkillAutocomplete(editor, "run a "); + editor.handleInput("\r"); + + expect(submitted).toBe("run a /skill:security-scan"); + expect(editor.getText()).toBe(""); + }); + + it("hides mid-prompt skill autocomplete immediately when Backspace removes the slash", async () => { + const editor = createSkillEditor(); + + await openMidPromptSkillAutocomplete(editor, "run a "); + editor.handleInput("\x7f"); + + expect(editor.getText()).toBe("run a "); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + it("opens mid-prompt skill autocomplete and inserts the skill token without wiping the draft on Tab", async () => { const editor = new Editor(defaultEditorTheme); editor.setAutocompleteProvider( From c8df93ca31646b5edc3396e0b7d478c281031f63 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 17:23:00 +0000 Subject: [PATCH 18/91] fix(tools): preferred literal filesystem match over trailing :selector peel for read and grep splitPathAndSel unconditionally peels a trailing : chunk whenever it matches the read-tool selector grammar (raw, conflicts, N-M, N+K, ...). On POSIX, filenames may legitimately contain colons, so a real file named test:1-2 or log:raw was shredded to test/log before either read.ts or grep.parsePathSpecs stated anything and both surfaced "Path not found". Added splitPathAndSelPreferringLiteral(rawPath, cwd) alongside the strict splitter: it only overrides the peel when fs.stat succeeds against the raw path. Read (execute) and grep (parsePathSpecs) call the async variant for non-URL paths; internal-URL splitting stays unchanged. Regression covers splitter fallbacks, read/grep behavior on literal-colon files, and that :1-2 selectors still work when the base file is the only real match. Fixes #4618 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/tools/grep.ts | 10 +- packages/coding-agent/src/tools/path-utils.ts | 22 +++ packages/coding-agent/src/tools/read.ts | 3 +- .../tools/path-literal-colon-selector.test.ts | 166 ++++++++++++++++++ 5 files changed, 200 insertions(+), 5 deletions(-) create mode 100644 packages/coding-agent/test/tools/path-literal-colon-selector.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 46b71b427..52ff73bb2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `read` and `grep` refusing to access filesystem paths whose names end in a selector-shaped suffix (e.g. `test:1-2`, `log:raw`) by preferring a literal match over the trailing `:` peel when the raw path exists on disk ([#4618](https://github.com/can1357/oh-my-pi/issues/4618)). + ## [16.3.7] - 2026-07-05 ### Added diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index dbc5f3371..f77fb4633 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -53,7 +53,7 @@ import { resolveToolSearchScope, selectorLineRanges, splitInternalUrlSel, - splitPathAndSel, + splitPathAndSelPreferringLiteral, toPathList, } from "./path-utils"; import { @@ -147,7 +147,7 @@ function isReadSelectorGrammar(sel: string): boolean { return lower === "raw" || lower === "conflicts" || parseLineRanges(sel) !== null; } -function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] { +async function parsePathSpecs(rawEntries: readonly string[], cwd: string): Promise { const specs: GrepPathSpec[] = []; for (const entry of rawEntries) { // Internal URLs (`artifact://`, `skill://`, …) use the URL-aware splitter, @@ -168,7 +168,9 @@ function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] { specs.push({ original: entry, clean: internalSplit.path, ranges: selectorLineRanges(internalSplit.sel) }); continue; } - const split = splitPathAndSel(entry); + // Prefer a literal filesystem match when one exists — a real file named + // `test:1-2` outranks the `:1-2` selector interpretation (issue #4618). + const split = await splitPathAndSelPreferringLiteral(entry, cwd); let clean = entry; let ranges: [LineRange, ...LineRange[]] | undefined; if (split.sel) { @@ -897,7 +899,7 @@ export class GrepTool implements AgentTool const scopedPaths = toPathList(rawPath); const effectivePaths = scopedPaths.length > 0 ? scopedPaths : ["."]; const rawEntries = await expandDelimitedPathEntries(effectivePaths, this.session.cwd); - const pathSpecs = parsePathSpecs(rawEntries); + const pathSpecs = await parsePathSpecs(rawEntries, this.session.cwd); const paths = pathSpecs.map(spec => spec.clean); const materializedExternalPaths = new Map(); const materializeExternalUrlForSearch = async (rawPath: string) => { diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index b0e51eb3a..0d2075358 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -315,6 +315,28 @@ export function splitPathAndSel(rawPath: string): { path: string; sel?: string } return { path: basePath, sel }; } +/** + * Async sibling of {@link splitPathAndSel} that prefers a literal filesystem + * path over selector interpretation when the raw input exists on disk. + * Filenames whose tail matches the selector grammar (e.g. `test:1-2`, `log:raw`) + * are legal on POSIX; without this the strict splitter peels the tail and both + * `read` and `grep` refuse to open the real file (see issue #4618). Mirrors + * {@link parseSearchPathPreferringLiteral} for glob-shaped literal paths. + */ +export async function splitPathAndSelPreferringLiteral( + rawPath: string, + cwd: string, +): Promise<{ path: string; sel?: string }> { + const strict = splitPathAndSel(rawPath); + if (strict.sel === undefined) return strict; + try { + await fs.promises.stat(resolveToCwd(rawPath, cwd)); + return { path: rawPath }; + } catch { + return strict; + } +} + /** * Variant of {@link splitPathAndSel} for internal URLs (`scheme://...`). * diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index cd838ca42..47b1a4cbe 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -103,6 +103,7 @@ import { splitDelimitedPathEntry, splitInternalUrlSel, splitPathAndSel, + splitPathAndSelPreferringLiteral, } from "./path-utils"; import { formatBytes, replaceTabs, shortenPath, wrapBrackets } from "./render-utils"; import { @@ -2246,7 +2247,7 @@ export class ReadTool implements AgentTool { ); } - const localTarget = splitPathAndSel(readPath); + const localTarget = await splitPathAndSelPreferringLiteral(readPath, this.session.cwd); const localReadPath = localTarget.path; const parsed = parseSel(localTarget.sel); diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts new file mode 100644 index 000000000..50ee5f159 --- /dev/null +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -0,0 +1,166 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { splitPathAndSel, splitPathAndSelPreferringLiteral } from "@oh-my-pi/pi-coding-agent/tools/path-utils"; +import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; +import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import { GrepTool } from "../../src/tools/grep"; + +function getText(result: { content: Array<{ type: string; text?: string }> }): string { + return result.content + .filter(entry => entry.type === "text") + .map(entry => entry.text ?? "") + .join("\n"); +} + +// Regression: filenames whose tail matches the read-tool selector grammar +// (e.g. `test:1-2`, `log:raw`) used to be shredded by `splitPathAndSel` before +// either tool checked the filesystem — see issue #4618. Both `read` and `grep` +// must prefer a real literal file over the selector interpretation. +describe("literal colon filename resolution (issue #4618)", () => { + let tmpDir: string; + + beforeEach(async () => { + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "literal-colon-")); + }); + + afterEach(async () => { + await removeWithRetries(tmpDir); + }); + + function createSession(overrides: Partial = {}): ToolSession { + return { + cwd: tmpDir, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "grep.contextBefore": 0, "grep.contextAfter": 0 }), + ...overrides, + }; + } + + describe("splitPathAndSelPreferringLiteral", () => { + it("keeps the raw path intact when a literal colon file exists on disk", async () => { + const literal = "test:1-2"; + await Bun.write(path.join(tmpDir, literal), "test\n"); + + // Strict splitter still peels — this documents the contract the + // literal-preferring variant sits on top of. + expect(splitPathAndSel(literal)).toEqual({ path: "test", sel: "1-2" }); + + expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); + }); + + it("falls back to selector interpretation when the literal path does not exist", async () => { + // No file created — the selector split wins because the raw path + // cannot be stat'd. + expect(await splitPathAndSelPreferringLiteral("test:1-2", tmpDir)).toEqual({ + path: "test", + sel: "1-2", + }); + }); + + it("also protects `:raw`-shaped literal filenames", async () => { + const literal = "log:raw"; + await Bun.write(path.join(tmpDir, literal), "line one\nline two\n"); + expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); + }); + + it("returns the strict split unchanged when there is no selector tail", async () => { + expect(await splitPathAndSelPreferringLiteral("plain.txt", tmpDir)).toEqual({ + path: "plain.txt", + }); + }); + }); + + describe("read tool", () => { + it("reads a literal file whose name ends in a selector-shaped suffix", async () => { + const literal = "test:1-2"; + const absolute = path.join(tmpDir, literal); + await Bun.write(absolute, "test\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal", { path: absolute }); + const output = getText(result); + + expect(output).toContain("test"); + // The strict split would have opened `test` (which doesn't exist) + // and thrown "Path 'test' not found". + expect(output).not.toMatch(/not found/i); + }); + + it("prefers a real `foo:1-2` file over interpreting `:1-2` as a range on `foo`", async () => { + await Bun.write(path.join(tmpDir, "foo"), "line 1\nline 2\nline 3\n"); + await Bun.write(path.join(tmpDir, "foo:1-2"), "colon file wins\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal-wins", { + path: path.join(tmpDir, "foo:1-2"), + }); + const output = getText(result); + + expect(output).toContain("colon file wins"); + expect(output).not.toContain("line 1"); + }); + + it("still honors the `:5-10` selector when only the base file exists on disk", async () => { + const absolute = path.join(tmpDir, "notes"); + const lines = Array.from({ length: 40 }, (_, i) => `line ${i + 1}`).join("\n"); + await Bun.write(absolute, `${lines}\n`); + + const session = createSession(); + session.settings.set("read.summarize.enabled", false); + const tool = new ReadTool(session); + const result = await tool.execute("read-selector-preserved", { + path: `${absolute}:5-10`, + }); + const output = getText(result); + + expect(output).toContain("line 5"); + expect(output).toContain("line 10"); + // Lines well outside the requested range must not appear — the selector + // still peels because the raw `notes:5-10` path does not exist literally. + expect(output).not.toContain("line 30"); + expect(output).not.toContain("line 40"); + }); + }); + + describe("grep tool", () => { + it("searches inside a literal `test:1-2` file", async () => { + const literal = "test:1-2"; + const absolute = path.join(tmpDir, literal); + await Bun.write(absolute, "needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-literal", { + pattern: "needle", + path: absolute, + }); + const output = getText(result); + + expect(output).toContain("needle"); + expect(output).not.toMatch(/not found/i); + }); + + it("preserves `:N-M` line-range filtering when the literal file does not exist", async () => { + const absolute = path.join(tmpDir, "notes.txt"); + await Bun.write(absolute, "one\ntwo\nthree\nfour\n"); + + const tool = new GrepTool(createSession()); + const rangedResult = await tool.execute("grep-range-filter", { + pattern: ".", + path: `${absolute}:1-2`, + }); + const rangedOutput = getText(rangedResult); + + expect(rangedOutput).toContain("one"); + expect(rangedOutput).toContain("two"); + // Lines outside the range are filtered out. + expect(rangedOutput).not.toContain("three"); + expect(rangedOutput).not.toContain("four"); + }); + }); +}); From 48a6e46750fb9ccd1b419c1c77ad5758caa6b0fc Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 17:33:21 +0000 Subject: [PATCH 19/91] fix(tools): hoisted literal-preferring split ahead of archive/sqlite/pdf dispatch in read The prior hunk placed the literal-preferring split after resolveArchiveReadPath, resolveSqliteReadPath, and splitPdfImageMemberReadPath, so a real POSIX file such as data.zip:1-2 or notes.db:1-2 still got hijacked: the archive/sqlite resolvers matched the base extension, opened data.zip / notes.db, and errored on the phantom :1-2 member before the literal file was ever considered. Now the async splitter runs first. When the strict grammar would have peeled a suffix but the literal path stats successfully, all three structured dispatchers decline. Otherwise the ordering is unchanged, so archive/sqlite/ pdf-image reads keep working when the literal file does not exist. Regression coverage adds `data.zip:1-2` and `notes.db:1-2` cases where the base archive/sqlite file also exists on disk, exercising the exact ordering bug the reviewer flagged. --- packages/coding-agent/src/tools/read.ts | 90 +++++++++++-------- .../tools/path-literal-colon-selector.test.ts | 40 +++++++++ 2 files changed, 92 insertions(+), 38 deletions(-) diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 47b1a4cbe..643c1f175 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -2206,48 +2206,62 @@ export class ReadTool implements AgentTool { // resolution share misses instead of re-globbing the workspace. const suffixCache: SuffixMatchCache = new Map(); - const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal); - if (archivePath) { - const archiveSubPath = splitPathAndSel(archivePath.archiveSubPath); - const archiveParsed = parseSel(archiveSubPath.sel); - return this.#readArchive( - readPath, - archiveParsed, - { ...archivePath, archiveSubPath: archiveSubPath.path }, - signal, - ); - } + // Prefer a literal filesystem match over the strict `:` peel so real + // POSIX filenames whose tail matches the selector grammar (e.g. `test:1-2`, + // `data.zip:1-2`, `notes.db:raw`) win over the structured-path resolvers + // below. When the raw path resolves literally on disk AND the strict + // splitter would have peeled a selector, the archive / sqlite / pdf-image + // dispatchers must decline — otherwise `data.zip:1-2` still opens + // `data.zip` and errors on the phantom member (issue #4618). + const literalSplit = await splitPathAndSelPreferringLiteral(readPath, this.session.cwd); + // Literal wins whenever the strict grammar would have peeled a suffix but + // the async splitter decided to keep the raw path (fs.stat succeeded). + const rawPathIsLiteral = literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined; - const sqlitePath = await this.#resolveSqliteReadPath(readPath, suffixCache, signal); - if (sqlitePath) { - return this.#readSqlite(sqlitePath, signal); - } - - const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath); - if (pdfImageMemberPath) { - let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd); - let suffixResolution: { from: string; to: string } | undefined; - try { - const stat = await Bun.file(absolutePdfPath).stat(); - if (stat.isDirectory()) - throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`); - } catch (error) { - if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error; - const suffixMatch = await this.#findSuffixMatchCached(suffixCache, pdfImageMemberPath.pdfPath, signal); - if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`); - absolutePdfPath = suffixMatch.absolutePath; - suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath }; + if (!rawPathIsLiteral) { + const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal); + if (archivePath) { + const archiveSubPath = splitPathAndSel(archivePath.archiveSubPath); + const archiveParsed = parseSel(archiveSubPath.sel); + return this.#readArchive( + readPath, + archiveParsed, + { ...archivePath, archiveSubPath: archiveSubPath.path }, + signal, + ); + } + + const sqlitePath = await this.#resolveSqliteReadPath(readPath, suffixCache, signal); + if (sqlitePath) { + return this.#readSqlite(sqlitePath, signal); + } + + const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath); + if (pdfImageMemberPath) { + let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd); + let suffixResolution: { from: string; to: string } | undefined; + try { + const stat = await Bun.file(absolutePdfPath).stat(); + if (stat.isDirectory()) + throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`); + } catch (error) { + if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error; + const suffixMatch = await this.#findSuffixMatchCached(suffixCache, pdfImageMemberPath.pdfPath, signal); + if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`); + absolutePdfPath = suffixMatch.absolutePath; + suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath }; + } + return this.#readPdfImageMember( + absolutePdfPath, + pdfImageMemberPath.pdfPath, + pdfImageMemberPath.member, + suffixResolution, + signal, + ); } - return this.#readPdfImageMember( - absolutePdfPath, - pdfImageMemberPath.pdfPath, - pdfImageMemberPath.member, - suffixResolution, - signal, - ); } - const localTarget = await splitPathAndSelPreferringLiteral(readPath, this.session.cwd); + const localTarget = literalSplit; const localReadPath = localTarget.path; const parsed = parseSel(localTarget.sel); diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts index 50ee5f159..f56b16d2d 100644 --- a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -126,6 +126,46 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).not.toContain("line 30"); expect(output).not.toContain("line 40"); }); + + it("reads a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { + // A real POSIX file whose name ends in a selector-shaped tail after an + // archive extension. The archive resolver would otherwise open `data.zip` + // alongside it and error on the phantom member. + const baseArchive = path.join(tmpDir, "data.zip"); + // Empty zip bytes — the file just needs to stat as a real archive so + // the archive resolver would happily accept it. + await Bun.write( + baseArchive, + new Uint8Array([0x50, 0x4b, 0x05, 0x06, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]), + ); + const literal = path.join(tmpDir, "data.zip:1-2"); + await Bun.write(literal, "literal archive-shaped file\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal-zip-selector", { path: literal }); + const output = getText(result); + + expect(output).toContain("literal archive-shaped file"); + }); + + it("reads a literal file that looks like a sqlite selector (`notes.db:1-2`)", async () => { + // A real POSIX file whose base name matches a sqlite-shaped path plus a + // selector-shaped tail. The sqlite resolver would misroute this to + // `notes.db` and try to open a table named `1-2`. + const baseDb = path.join(tmpDir, "notes.db"); + // SQLite database header (16-byte magic string plus zero-padding). + const header = new Uint8Array(4096); + header.set(Buffer.from("SQLite format 3\0", "utf-8"), 0); + await Bun.write(baseDb, header); + const literal = path.join(tmpDir, "notes.db:1-2"); + await Bun.write(literal, "literal db-shaped file\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal-db-selector", { path: literal }); + const output = getText(result); + + expect(output).toContain("literal db-shaped file"); + }); }); describe("grep tool", () => { From c40b0ff5daa686567218e6ae28234cf95eaa6fed Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 17:38:20 +0000 Subject: [PATCH 20/91] fix(tools): skipped grep archive materialization for literal filesystem matches parsePathSpecs preserved an existing literal path like data.zip:1-2, but resolveArchiveSearchPaths only received the cleaned path strings and reparsed the same literal as archive data.zip plus member 1-2. If data.zip existed, grep materialized or errored on the archive member before searching the literal file. GrepPathSpec now carries whether a local entry was kept because the raw filesystem path exists. Archive materialization consumes the specs instead of bare strings and skips those literal matches, while ordinary archive selectors still materialize as before. Regression coverage adds grep over data.zip:1-2 with a real data.zip alongside, proving the literal file is searched instead of the archive member. --- packages/coding-agent/src/tools/grep.ts | 20 +++++++++------ .../tools/path-literal-colon-selector.test.ts | 25 ++++++++++++++++--- 2 files changed, 33 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index f77fb4633..53f677c48 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -53,6 +53,7 @@ import { resolveToolSearchScope, selectorLineRanges, splitInternalUrlSel, + splitPathAndSel, splitPathAndSelPreferringLiteral, toPathList, } from "./path-utils"; @@ -119,6 +120,7 @@ const SEARCH_GREP_TIMEOUT_MS = 30_000; interface GrepPathSpec { original: string; clean: string; + literalFilesystemMatch?: boolean; ranges?: [LineRange, ...LineRange[]]; } @@ -170,7 +172,9 @@ async function parsePathSpecs(rawEntries: readonly string[], cwd: string): Promi } // Prefer a literal filesystem match when one exists — a real file named // `test:1-2` outranks the `:1-2` selector interpretation (issue #4618). + const strictSplit = splitPathAndSel(entry); const split = await splitPathAndSelPreferringLiteral(entry, cwd); + const literalFilesystemMatch = strictSplit.sel !== undefined && split.sel === undefined; let clean = entry; let ranges: [LineRange, ...LineRange[]] | undefined; if (split.sel) { @@ -186,7 +190,7 @@ async function parsePathSpecs(rawEntries: readonly string[], cwd: string): Promi clean = split.path; ranges = parsed; } - specs.push({ original: entry, clean, ranges }); + specs.push({ original: entry, clean, literalFilesystemMatch, ranges }); } return specs; } @@ -222,7 +226,7 @@ function matchAbsolutePath(matchPath: string, searchPath: string): string { * cleanup hook the caller MUST invoke in a `finally`. */ async function resolveArchiveSearchPaths( - paths: string[], + pathSpecs: readonly GrepPathSpec[], cwd: string, ): Promise<{ resolvedPaths: string[]; @@ -231,17 +235,18 @@ async function resolveArchiveSearchPaths( unreadable: string[]; cleanup: () => Promise; }> { - const resolvedPaths = paths.slice(); + const resolvedPaths = pathSpecs.map(spec => spec.clean); const displayMap = new Map(); const displaySet = new Set(); const unreadable: string[] = []; let tempDir: string | undefined; const archiveCache = new Map(); - for (let idx = 0; idx < paths.length; idx++) { - const entry = paths[idx]; + for (let idx = 0; idx < pathSpecs.length; idx++) { + const spec = pathSpecs[idx]; + if (!spec || spec.literalFilesystemMatch) continue; + const entry = spec.clean; const candidates = parseArchivePathCandidates(entry); - // Longest archive prefix first; we want the one whose member portion is non-empty. const member = candidates.find(c => c.subPath !== "" && c.archivePath !== entry); if (!member) continue; @@ -900,7 +905,6 @@ export class GrepTool implements AgentTool const effectivePaths = scopedPaths.length > 0 ? scopedPaths : ["."]; const rawEntries = await expandDelimitedPathEntries(effectivePaths, this.session.cwd); const pathSpecs = await parsePathSpecs(rawEntries, this.session.cwd); - const paths = pathSpecs.map(spec => spec.clean); const materializedExternalPaths = new Map(); const materializeExternalUrlForSearch = async (rawPath: string) => { const target = parseReadUrlTarget(rawPath); @@ -919,7 +923,7 @@ export class GrepTool implements AgentTool displaySet: archiveDisplaySet, unreadable: archiveUnreadable, cleanup: cleanupArchiveScratch, - } = await resolveArchiveSearchPaths(paths, this.session.cwd); + } = await resolveArchiveSearchPaths(pathSpecs, this.session.cwd); try { const internalResolution = await resolveInternalSearchInputs({ pathSpecs, diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts index f56b16d2d..db93ae383 100644 --- a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -16,6 +16,8 @@ function getText(result: { content: Array<{ type: string; text?: string }> }): s .join("\n"); } +const EMPTY_ZIP_EOCD = new Uint8Array([0x50, 0x4b, 0x05, 0x06, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]); + // Regression: filenames whose tail matches the read-tool selector grammar // (e.g. `test:1-2`, `log:raw`) used to be shredded by `splitPathAndSel` before // either tool checked the filesystem — see issue #4618. Both `read` and `grep` @@ -134,10 +136,7 @@ describe("literal colon filename resolution (issue #4618)", () => { const baseArchive = path.join(tmpDir, "data.zip"); // Empty zip bytes — the file just needs to stat as a real archive so // the archive resolver would happily accept it. - await Bun.write( - baseArchive, - new Uint8Array([0x50, 0x4b, 0x05, 0x06, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]), - ); + await Bun.write(baseArchive, EMPTY_ZIP_EOCD); const literal = path.join(tmpDir, "data.zip:1-2"); await Bun.write(literal, "literal archive-shaped file\n"); @@ -185,6 +184,24 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).not.toMatch(/not found/i); }); + it("searches a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { + // The base archive exists too; grep must not rematerialize the raw + // literal path as archive `data.zip` plus phantom member `1-2`. + const baseArchive = path.join(tmpDir, "data.zip"); + await Bun.write(baseArchive, EMPTY_ZIP_EOCD); + const literal = path.join(tmpDir, "data.zip:1-2"); + await Bun.write(literal, "literal archive needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-literal-zip-selector", { + pattern: "needle", + path: literal, + }); + const output = getText(result); + + expect(output).toContain("literal archive needle"); + }); + it("preserves `:N-M` line-range filtering when the literal file does not exist", async () => { const absolute = path.join(tmpDir, "notes.txt"); await Bun.write(absolute, "one\ntwo\nthree\nfour\n"); From c493d12f95d1178d69f7464131a06f59c324a7b2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 17:44:37 +0000 Subject: [PATCH 21/91] fix(tools): preserved escaped literal selector-shaped paths splitPathAndSelPreferringLiteral only statted resolveToCwd(rawPath), so shell-escaped paths such as dir/a\ b:1-2 missed the existing dir/a b:1-2 file and fell back to the strict selector peel. That let read target dir/a\ b with a range instead of the literal filename. The helper now probes resolveReadPath(rawPath, cwd), reusing the read path resolver's existing escaped-space and filesystem variant normalization before deciding whether the literal file exists. Grep also stores the resolved filesystem path for literal matches so the later search-scope parser does not reinterpret backslashes as path separators. Regression coverage now includes helper, read, and grep cases for dir/a\ b:1-2. --- packages/coding-agent/src/tools/grep.ts | 4 +-- packages/coding-agent/src/tools/path-utils.ts | 2 +- .../tools/path-literal-colon-selector.test.ts | 34 +++++++++++++++++++ 3 files changed, 37 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index 53f677c48..c459a682b 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -175,9 +175,9 @@ async function parsePathSpecs(rawEntries: readonly string[], cwd: string): Promi const strictSplit = splitPathAndSel(entry); const split = await splitPathAndSelPreferringLiteral(entry, cwd); const literalFilesystemMatch = strictSplit.sel !== undefined && split.sel === undefined; - let clean = entry; + let clean = literalFilesystemMatch ? resolveReadPath(entry, cwd) : entry; let ranges: [LineRange, ...LineRange[]] | undefined; - if (split.sel) { + if (!literalFilesystemMatch && split.sel) { const parsed = parseLineRanges(split.sel); if (!parsed) { throw new ToolError( diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index 0d2075358..70c1855a6 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -330,7 +330,7 @@ export async function splitPathAndSelPreferringLiteral( const strict = splitPathAndSel(rawPath); if (strict.sel === undefined) return strict; try { - await fs.promises.stat(resolveToCwd(rawPath, cwd)); + await fs.promises.stat(resolveReadPath(rawPath, cwd)); return { path: rawPath }; } catch { return strict; diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts index db93ae383..0db58325e 100644 --- a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -56,6 +56,15 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); }); + it("keeps a shell-escaped literal path intact when the resolved file exists", async () => { + await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); + await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal\n"); + + expect(await splitPathAndSelPreferringLiteral("dir/a\\ b:1-2", tmpDir)).toEqual({ + path: "dir/a\\ b:1-2", + }); + }); + it("falls back to selector interpretation when the literal path does not exist", async () => { // No file created — the selector split wins because the raw path // cannot be stat'd. @@ -94,6 +103,17 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).not.toMatch(/not found/i); }); + it("reads a shell-escaped literal file whose name ends in a selector-shaped suffix", async () => { + await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); + await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal read\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-escaped-literal", { path: "dir/a\\ b:1-2" }); + const output = getText(result); + + expect(output).toContain("escaped literal read"); + }); + it("prefers a real `foo:1-2` file over interpreting `:1-2` as a range on `foo`", async () => { await Bun.write(path.join(tmpDir, "foo"), "line 1\nline 2\nline 3\n"); await Bun.write(path.join(tmpDir, "foo:1-2"), "colon file wins\n"); @@ -184,6 +204,20 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).not.toMatch(/not found/i); }); + it("searches a shell-escaped literal file whose name ends in a selector-shaped suffix", async () => { + await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); + await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-escaped-literal", { + pattern: "needle", + path: "dir/a\\ b:1-2", + }); + const output = getText(result); + + expect(output).toContain("escaped literal needle"); + }); + it("searches a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { // The base archive exists too; grep must not rematerialize the raw // literal path as archive `data.zip` plus phantom member `1-2`. From ff3b0c795c6e025e0febaf19328852028d126163 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 18:09:29 +0000 Subject: [PATCH 22/91] fix(tools): added explicit selector fields for read and grep The literal-path stat fallback made selector-shaped filenames accessible, but it did not give callers a deterministic way to read or grep a range from a literal filename such as test:1-2. Encoding that as test:1-2:1-2 remained recursively ambiguous if a longer literal file later appeared. Read now accepts an optional selector field that is parsed independently from path. When selector is present, path is treated as the exact path first, so { path: "test:1-2", selector: "1-2" } always means lines 1-2 from the literal file test:1-2. Inline : remains supported for compatibility. Grep now accepts an optional line-range selector field with the same literal-path behavior. Explicit selectors bypass path suffix peeling, while archive/internal/URL routing still handles non-literal structured paths. Updated read/grep tool prompts and added deterministic regressions proving that a longer literal file like test:1-2:5-6 or test:1-2:2-2 does not change the meaning of { path: "test:1-2", selector: ... }. --- .../coding-agent/src/prompts/tools/grep.md | 2 +- .../coding-agent/src/prompts/tools/read.md | 7 +- packages/coding-agent/src/tools/grep.ts | 36 +++++++- packages/coding-agent/src/tools/path-utils.ts | 21 +++-- packages/coding-agent/src/tools/read.ts | 87 +++++++++++++------ .../tools/path-literal-colon-selector.test.ts | 41 +++++++++ 6 files changed, 156 insertions(+), 38 deletions(-) diff --git a/packages/coding-agent/src/prompts/tools/grep.md b/packages/coding-agent/src/prompts/tools/grep.md index 78467d100..eef17d10e 100644 --- a/packages/coding-agent/src/prompts/tools/grep.md +++ b/packages/coding-agent/src/prompts/tools/grep.md @@ -2,7 +2,7 @@ Greps files using regex. - Rust regex (RE2-style): alternation is `foo|bar`, not GNU BRE-style `foo\|bar`; Rust word boundaries like `\bword\b` are supported. Use line anchors or post-filters instead of lookaround/backreferences. -- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). +- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). Literal colon filename + line range? Use `selector` (e.g. `{"path":"test:1-2","selector":"1-2"}`), not recursive `path:"test:1-2:1-2"`. - Cross-line patterns detected from literal `\n` or `\\n` in `pattern`. diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md index 554dc4cab..a54242eff 100644 --- a/packages/coding-agent/src/prompts/tools/read.md +++ b/packages/coding-agent/src/prompts/tools/read.md @@ -1,4 +1,4 @@ -Read files, directories, archives, SQLite, images, documents, internal resources, and web URLs via one `path`. +Read files, directories, archives, SQLite, images, documents, internal resources, and web URLs via `path` plus optional `selector`. - SHOULD parallelize independent reads. @@ -7,7 +7,8 @@ Read files, directories, archives, SQLite, images, documents, internal resources ## Parameters -- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Append `:` for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`). +- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Inline `:` still works for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`). +- `selector` — optional selector without leading `:` (e.g. `"50-200"`, `"raw"`, `"raw:50-100"`, `"conflicts"`). Use when `path` contains literal colons: `{"path":"test:1-2","selector":"1-2"}`. ## Selectors @@ -72,6 +73,6 @@ All URI schemes take the same line selectors. `artifact://` recovers spilled `ssh://host/` reads a remote text file (UTF-8, ≤1 MiB) or lists a directory one level deep, on a pre-configured SSH host or `~/.ssh/config` alias; `ssh://host/` lists the remote root and bare `ssh://` lists the configured hosts. Files are also writable via `write` and searchable via `search`; a directory only lists (`search` refuses a directory, `write` refuses to overwrite one). A literal `:`, `?`, or `#` in the remote path must be percent-encoded (`%3A`/`%3F`/`%23`) — a trailing `:sel` is read as a line selector, and `?`/`#` start a URL query/fragment. Requires a POSIX login shell (`sh`/`bash`/`zsh`); a Windows host or a non-POSIX shell (fish, csh/tcsh) is rejected — use the `ssh` tool there. -- Line ranges go in the selector: `path="src/foo.ts:50-200"`. +- Literal colon filename + selector? Use `selector`, not recursive `path:"file:sel:sel"`. - Summary footer names elided ranges? Re-issue ONLY those ranges. NEVER guess `..`/`…` content. diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index c459a682b..76d90f84c 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -49,6 +49,7 @@ import { parseLineRanges, pathTargetsSsh, type ResolvedSearchTarget, + resolveExistingReadPath, resolveReadPath, resolveToolSearchScope, selectorLineRanges, @@ -78,6 +79,9 @@ const searchSchema = type({ "path?": searchPathEntry.describe( 'file, directory, glob, internal URL, or ":" selector to search; pass several as a semicolon-delimited list ("src; tests"). Omitted -> searches the workspace root (".")', ), + "selector?": type("string").describe( + 'line selector without a leading colon (e.g. "50-100", "50+10", "50-100,200-300"); keeps `path` literal when filenames contain colons', + ), "case?": type("boolean").describe("case-sensitive search"), "gitignore?": type("boolean").describe("respect gitignore"), "skip?": type("number") @@ -149,9 +153,35 @@ function isReadSelectorGrammar(sel: string): boolean { return lower === "raw" || lower === "conflicts" || parseLineRanges(sel) !== null; } -async function parsePathSpecs(rawEntries: readonly string[], cwd: string): Promise { +async function parsePathSpecs( + rawEntries: readonly string[], + cwd: string, + explicitSelector?: string, +): Promise { + const explicitRanges = + explicitSelector === undefined || explicitSelector.length === 0 ? undefined : parseLineRanges(explicitSelector); + if (explicitSelector !== undefined && !explicitRanges) { + throw new ToolError( + `selector "${explicitSelector}" is invalid — use line ranges like "50-100", "50+10", or "50-100,200-300" without a leading colon`, + ); + } const specs: GrepPathSpec[] = []; for (const entry of rawEntries) { + if (explicitRanges) { + // Separate selector parameter makes `path` deterministic: first try the + // exact local filesystem path (with read-path normalization), then let + // archive/internal/URL resolution handle non-literal structured paths. + const localPath = /^[a-z][a-z0-9+.-]*:\/\//i.test(entry) + ? undefined + : await resolveExistingReadPath(entry, cwd); + specs.push({ + original: entry, + clean: localPath ?? entry, + literalFilesystemMatch: localPath !== undefined, + ranges: explicitRanges, + }); + continue; + } // Internal URLs (`artifact://`, `skill://`, …) use the URL-aware splitter, // which peels selector-shaped tails only for selector-capable schemes and // leaves opaque ones (`mcp://`) intact. Unlike filesystem paths, their @@ -886,7 +916,7 @@ export class GrepTool implements AgentTool _onUpdate?: AgentToolUpdateCallback, _toolContext?: AgentToolContext, ): Promise> { - const { pattern, path: rawPath, case: caseSensitive, gitignore, skip } = params; + const { pattern, path: rawPath, selector, case: caseSensitive, gitignore, skip } = params; return untilAborted(signal, async () => { // Preserve the pattern verbatim — leading/trailing whitespace is @@ -904,7 +934,7 @@ export class GrepTool implements AgentTool const scopedPaths = toPathList(rawPath); const effectivePaths = scopedPaths.length > 0 ? scopedPaths : ["."]; const rawEntries = await expandDelimitedPathEntries(effectivePaths, this.session.cwd); - const pathSpecs = await parsePathSpecs(rawEntries, this.session.cwd); + const pathSpecs = await parsePathSpecs(rawEntries, this.session.cwd, selector); const materializedExternalPaths = new Map(); const materializeExternalUrlForSearch = async (rawPath: string) => { const target = parseReadUrlTarget(rawPath); diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index 70c1855a6..807abc060 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -315,6 +315,18 @@ export function splitPathAndSel(rawPath: string): { path: string; sel?: string } return { path: basePath, sel }; } +/** Resolve a read-tool path variant and return it only when it already exists on disk. */ +export async function resolveExistingReadPath(filePath: string, cwd: string): Promise { + const resolved = resolveReadPath(filePath, cwd); + try { + await fs.promises.stat(resolved); + return resolved; + } catch (err) { + if (isEnoent(err) || isEnotdir(err)) return undefined; + return resolved; + } +} + /** * Async sibling of {@link splitPathAndSel} that prefers a literal filesystem * path over selector interpretation when the raw input exists on disk. @@ -329,12 +341,9 @@ export async function splitPathAndSelPreferringLiteral( ): Promise<{ path: string; sel?: string }> { const strict = splitPathAndSel(rawPath); if (strict.sel === undefined) return strict; - try { - await fs.promises.stat(resolveReadPath(rawPath, cwd)); - return { path: rawPath }; - } catch { - return strict; - } + const resolved = await resolveExistingReadPath(rawPath, cwd); + if (resolved !== undefined) return { path: rawPath }; + return strict; } /** diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 643c1f175..2412358b1 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -99,6 +99,7 @@ import { type LineRange, parseLineRanges, pathTargetsSsh, + resolveExistingReadPath, resolveReadPath, splitDelimitedPathEntry, splitInternalUrlSel, @@ -747,7 +748,10 @@ function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; membe const readSchema = type({ path: type("string").describe( - 'Local path, internal URI (e.g. "omp://", "issue://123", "pr://123"), or URL; append : for line ranges or raw mode (e.g. "src/foo.ts:50-100")', + 'Local path, internal URI (e.g. "omp://", "issue://123", "pr://123"), or URL. Inline : is still accepted for compatibility.', + ), + "selector?": type("string").describe( + 'selector without a leading colon (e.g. "50-100", "raw", "raw:50-100", "conflicts"); keeps `path` literal when filenames contain colons', ), }); @@ -2114,6 +2118,14 @@ export class ReadTool implements AgentTool { _toolContext?: AgentToolContext, ): Promise> { let { path: readPath } = params; + const explicitSelector = params.selector?.trim(); + const explicitParsedSelector = explicitSelector === undefined ? undefined : parseSel(explicitSelector); + if ( + params.selector !== undefined && + (explicitSelector === undefined || explicitSelector.length === 0 || explicitParsedSelector?.kind === "none") + ) { + throw invalidSelector(params.selector); + } if (readPath.startsWith("file://")) { readPath = expandPath(readPath); } @@ -2134,40 +2146,55 @@ export class ReadTool implements AgentTool { if (!this.session.settings.get("fetch.enabled")) { throw new ToolError("URL reads are disabled by settings."); } - if (parsedUrlTarget.ranges !== undefined) { + if (explicitParsedSelector?.kind === "conflicts") { + throw new ToolError("The explicit read selector `conflicts` is only supported for local files."); + } + const urlRaw = + explicitParsedSelector === undefined ? parsedUrlTarget.raw : isRawSelector(explicitParsedSelector); + const urlRanges = + explicitParsedSelector?.kind === "lines" ? explicitParsedSelector.ranges : parsedUrlTarget.ranges; + if (urlRanges !== undefined && urlRanges.length > 1) { const cached = await loadReadUrlCacheEntry( this.session, - { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, + { path: parsedUrlTarget.path, raw: urlRaw }, signal, { ensureArtifact: true, preferCached: true }, ); - return this.#buildInMemoryMultiRangeResult(cached.output, parsedUrlTarget.ranges, { + return this.#buildInMemoryMultiRangeResult(cached.output, urlRanges, { details: { ...cached.details }, sourceUrl: cached.details.finalUrl, entityLabel: "URL output", - raw: parsedUrlTarget.raw, + raw: urlRaw, immutable: true, }); } - if (parsedUrlTarget.offset !== undefined || parsedUrlTarget.limit !== undefined) { + const urlRange = urlRanges?.[0]; + const urlOffset = explicitParsedSelector?.kind === "lines" ? urlRange?.startLine : parsedUrlTarget.offset; + const urlLimit = + explicitParsedSelector?.kind === "lines" && urlRange + ? urlRange.endLine !== undefined + ? urlRange.endLine - urlRange.startLine + 1 + : undefined + : parsedUrlTarget.limit; + if (urlOffset !== undefined || urlLimit !== undefined) { const cached = await loadReadUrlCacheEntry( this.session, - { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, + { path: parsedUrlTarget.path, raw: urlRaw }, signal, { ensureArtifact: true, preferCached: true, }, ); - return this.#buildInMemoryTextResult(cached.output, parsedUrlTarget.offset, parsedUrlTarget.limit, { + return this.#buildInMemoryTextResult(cached.output, urlOffset, urlLimit, { details: { ...cached.details }, sourceUrl: cached.details.finalUrl, entityLabel: "URL output", - raw: parsedUrlTarget.raw, + raw: urlRaw, immutable: true, }); } - return executeReadUrl(this.session, { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, signal); + return executeReadUrl(this.session, { path: parsedUrlTarget.path, raw: urlRaw }, signal); } // Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://, omp://, issue://, pr://). @@ -2175,8 +2202,9 @@ export class ReadTool implements AgentTool { // off the URL and surfaced via parseSel rather than confusing handlers. const internalRouter = InternalUrlRouter.instance(); if (internalRouter.canHandle(readPath)) { - const internalTarget = splitInternalUrlSel(readPath); - const parsed = parseSel(internalTarget.sel); + const internalTarget = + explicitSelector === undefined ? splitInternalUrlSel(readPath) : { path: readPath, sel: explicitSelector }; + const parsed = explicitParsedSelector ?? parseSel(internalTarget.sel); if (internalTarget.sel !== undefined && parsed.kind === "none") { throw new ToolError( `Invalid selector ':${internalTarget.sel}' on '${internalTarget.path}'. Use :N, :N-M, :N+K, :N- (open-ended), a comma-separated list of ranges, :raw, or a range combined with raw (e.g. :raw:50-100).`, @@ -2193,7 +2221,10 @@ export class ReadTool implements AgentTool { skills: this.session.skills, }); if (localFile) { - readPath = internalTarget.sel === undefined ? localFile.path : `${localFile.path}:${internalTarget.sel}`; + readPath = + explicitSelector !== undefined || internalTarget.sel === undefined + ? localFile.path + : `${localFile.path}:${internalTarget.sel}`; } else { return this.#handleInternalUrl(internalTarget.path, parsed, signal); } @@ -2206,22 +2237,28 @@ export class ReadTool implements AgentTool { // resolution share misses instead of re-globbing the workspace. const suffixCache: SuffixMatchCache = new Map(); - // Prefer a literal filesystem match over the strict `:` peel so real - // POSIX filenames whose tail matches the selector grammar (e.g. `test:1-2`, - // `data.zip:1-2`, `notes.db:raw`) win over the structured-path resolvers - // below. When the raw path resolves literally on disk AND the strict - // splitter would have peeled a selector, the archive / sqlite / pdf-image - // dispatchers must decline — otherwise `data.zip:1-2` still opens - // `data.zip` and errors on the phantom member (issue #4618). - const literalSplit = await splitPathAndSelPreferringLiteral(readPath, this.session.cwd); - // Literal wins whenever the strict grammar would have peeled a suffix but - // the async splitter decided to keep the raw path (fs.stat succeeded). - const rawPathIsLiteral = literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined; + // Prefer a literal filesystem match over selector interpretation so real + // POSIX filenames containing selector-looking suffixes win over structured + // archive / sqlite / pdf-image dispatch. With explicit `selector`, `path` + // is exact: `path: "test:1-2", selector: "1-2"` means "lines 1-2 from + // the literal file test:1-2", without recursively depending on whether a + // longer `test:1-2:1-2` filename also exists (issue #4618). + const literalSplit = + explicitSelector === undefined + ? await splitPathAndSelPreferringLiteral(readPath, this.session.cwd) + : { path: readPath, sel: explicitSelector }; + const rawPathIsLiteral = + explicitSelector !== undefined + ? readPath.includes(":") && (await resolveExistingReadPath(readPath, this.session.cwd)) !== undefined + : literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined; if (!rawPathIsLiteral) { const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal); if (archivePath) { - const archiveSubPath = splitPathAndSel(archivePath.archiveSubPath); + const archiveSubPath = + explicitSelector === undefined + ? splitPathAndSel(archivePath.archiveSubPath) + : { path: archivePath.archiveSubPath, sel: explicitSelector }; const archiveParsed = parseSel(archiveSubPath.sel); return this.#readArchive( readPath, diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts index 0db58325e..1792c1ca4 100644 --- a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -149,6 +149,28 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).not.toContain("line 40"); }); + it("uses explicit `selector` to read lines from a literal selector-shaped filename deterministically", async () => { + const literal = path.join(tmpDir, "test:1-2"); + const longerLiteral = path.join(tmpDir, "test:1-2:5-6"); + const lines = Array.from({ length: 40 }, (_, i) => `literal line ${i + 1}`).join("\n"); + await Bun.write(literal, `${lines}\n`); + await Bun.write(longerLiteral, "wrong longer literal\n"); + + const session = createSession(); + session.settings.set("read.summarize.enabled", false); + const tool = new ReadTool(session); + const result = await tool.execute("read-explicit-selector-literal", { + path: literal, + selector: "5-6", + }); + const output = getText(result); + + expect(output).toContain("literal line 5"); + expect(output).toContain("literal line 6"); + expect(output).not.toContain("literal line 30"); + expect(output).not.toContain("wrong longer literal"); + }); + it("reads a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { // A real POSIX file whose name ends in a selector-shaped tail after an // archive extension. The archive resolver would otherwise open `data.zip` @@ -204,6 +226,25 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).not.toMatch(/not found/i); }); + it("uses explicit `selector` to grep a literal selector-shaped filename deterministically", async () => { + const literal = path.join(tmpDir, "test:1-2"); + const longerLiteral = path.join(tmpDir, "test:1-2:2-2"); + await Bun.write(literal, "needle outside\nneedle inside\nneedle outside again\n"); + await Bun.write(longerLiteral, "wrong longer literal needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-explicit-selector-literal", { + pattern: "needle", + path: literal, + selector: "2-2", + }); + const output = getText(result); + + expect(output).toContain("needle inside"); + expect(output).not.toContain("needle outside again"); + expect(output).not.toContain("wrong longer literal"); + }); + it("searches a shell-escaped literal file whose name ends in a selector-shaped suffix", async () => { await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal needle\n"); From fc71df400fd6c2d020c2f6e2c8e6355543577f0e Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 18:14:32 +0000 Subject: [PATCH 23/91] fix(ai): recognized proxy stale-anchor codes as codex previous_response chain expiry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `isCodexStalePreviousResponseError` short-circuited for `CodexProviderStreamError` by comparing `error.code` only to `previous_response_not_found`, so a proxy code such as `codex_previous_response_stale` never reached the message-based fallback. WebSocket continuations that hit a stale upstream response anchor surfaced the terminal error to the user instead of retrying with full context. Codex WebSocket continuations now treat both the OpenAI-standard `previous_response_not_found` and the proxy `codex_previous_response_stale` code as the same recovery class, and every `Error` — not just plain ones — falls through to the existing `previous[ _]?response` / `expired|stale|...` message check. Fixes #4624 --- packages/ai/CHANGELOG.md | 4 + .../src/providers/openai-codex-responses.ts | 19 ++- packages/ai/test/openai-codex-stream.test.ts | 108 ++++++++++++++++++ 3 files changed, 127 insertions(+), 4 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d49636d82..ba1b5ca46 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI Codex WebSocket continuations to treat proxy stale-anchor codes such as `codex_previous_response_stale` as an expired `previous_response_id` chain — same recovery class as the OpenAI-standard `previous_response_not_found` — so the turn is retried with full context instead of surfacing the error to the user ([#4624](https://github.com/can1357/oh-my-pi/issues/4624)). + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 08ca0a434..abbbf6d53 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -1171,12 +1171,23 @@ function getOutputBlockStartEventType(block: CodexOutputBlock): "thinking_start" return "toolcall_start"; } +const CODEX_STALE_PREVIOUS_RESPONSE_CODES: Record = { + // OpenAI-standard code for an expired/missing `previous_response_id` chain. + previous_response_not_found: true, + // Proxy-specific: upstream response anchor expired. Same recovery class — + // retry the turn with full context and no `previous_response_id`. + codex_previous_response_stale: true, +}; + function isCodexStalePreviousResponseError(error: unknown): boolean { - if (error instanceof CodexProviderStreamError) return error.code === "previous_response_not_found"; if (!(error instanceof Error)) return false; - if ((error as { code?: string }).code === "previous_response_not_found") return true; - // "unsupported": the backend intermittently rejects the parameter outright - // with `{"detail":"Unsupported parameter: previous_response_id"}` (no + if ("code" in error && typeof error.code === "string" && CODEX_STALE_PREVIOUS_RESPONSE_CODES[error.code]) { + return true; + } + // Message-based fallback for providers/proxies that report the condition + // without a canonical code. Also covers "unsupported": the backend + // intermittently rejects the parameter outright with + // `{"detail":"Unsupported parameter: previous_response_id"}` (no // `error.code`); treat it like a stale chain so the turn replays with full // context instead of surfacing the 400. return ( diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index b7979ab40..439846d1f 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -2660,6 +2660,114 @@ describe("openai-codex streaming", () => { lastPreviousResponseId: undefined, }); }); + it("retries websocket continuations when a proxy reports a stale previous response anchor", async () => { + const tempDir = TempDir.createSync("@pi-codex-stream-"); + setAgentDir(tempDir.path()); + const token = createCodexTestToken(); + const sentRequests: Array> = []; + const fetchMock = vi.fn(async () => { + throw new Error("SSE fallback should not be called"); + }); + + class ProxyStaleAnchorWebSocket extends MockWebSocket { + constructor(url: string, options?: { headers?: WsHeaders }) { + super(url, options); + this.scheduleOpen(); + } + + send(data: string): void { + const request = JSON.parse(data) as Record; + sentRequests.push(request); + const requestIndex = sentRequests.length; + + if (requestIndex === 1) { + this.emitCodexResponse({ + messageId: "msg_1", + responseId: "resp_1", + text: "First answer", + terminalType: "response.completed", + includeCreated: true, + }); + return; + } + + if (requestIndex === 2) { + expect(request.previous_response_id).toBe("resp_1"); + this.sendJson({ + type: "error", + code: "codex_previous_response_stale", + message: "Upstream previous response anchor expired; retry without previous_response_id.", + }); + return; + } + + if (requestIndex === 3) { + expect(request.previous_response_id).toBeUndefined(); + this.emitCodexResponse({ + messageId: "msg_3", + responseId: "resp_3", + text: "Second answer", + terminalType: "response.completed", + includeCreated: true, + }); + return; + } + + throw new Error(`Unexpected websocket request index: ${requestIndex}`); + } + } + + global.WebSocket = ProxyStaleAnchorWebSocket as unknown as typeof WebSocket; + const model = createCodexTestModel("https://chatgpt.com/backend-api"); + const providerSessionState = new Map(); + const firstContext: Context = { + systemPrompt: ["You are a helpful assistant."], + messages: [{ role: "user", content: "First question", timestamp: Date.now() }], + }; + const firstResponse = await streamOpenAICodexResponses(model, firstContext, { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-proxy-stale-anchor-session", + providerSessionState, + }).result(); + const secondContext: Context = { + systemPrompt: ["You are a helpful assistant."], + messages: [ + ...firstContext.messages, + firstResponse, + { role: "user", content: "Second question", timestamp: Date.now() + 1 }, + ], + }; + + const secondResponse = await streamOpenAICodexResponses(model, secondContext, { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-proxy-stale-anchor-session", + providerSessionState, + }).result(); + + expect(secondResponse.stopReason).toBe("stop"); + expect(JSON.stringify(secondResponse.content)).toContain("Second answer"); + expect(fetchMock).not.toHaveBeenCalled(); + expect(sentRequests).toHaveLength(3); + expect(sentRequests[2]?.prompt_cache_key).toBe("ws-proxy-stale-anchor-session"); + const retryInput = sentRequests[2]?.input; + expect(Array.isArray(retryInput)).toBe(true); + expect(JSON.stringify(retryInput)).toContain("First question"); + expect(JSON.stringify(retryInput)).toContain("Second question"); + + const stats = getOpenAICodexWebSocketDebugStats(model, { + sessionId: "ws-proxy-stale-anchor-session", + providerSessionState, + }); + expect(stats).toEqual({ + fullContextRequests: 2, + deltaRequests: 1, + lastInputItems: (retryInput as unknown[]).length, + lastDeltaInputItems: undefined, + lastPreviousResponseId: undefined, + }); + }); it("uses websocket v2 beta header when v2 mode is enabled", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); From 6c8e7625a2fb4d739383d1917c50be682e51b785 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 18:17:59 +0000 Subject: [PATCH 24/91] fix(tools): tightened literal-path probe against stat ambiguity resolveExistingReadPath treated any stat failure other than ENOENT/ENOTDIR as "exists" and any other resolved path was considered a hit. That silently reinterpreted a real literal path such as test:1-2 as test plus selector 1-2 whenever the raw path was a dangling symlink, sat under an unreadable parent, or hit a transient I/O error. The new probeLiteralPathExists returns "exists" / "missing" / "unknown" from an lstat probe. splitPathAndSelPreferringLiteral now falls back to the strict selector split only on "missing"; both "exists" and "unknown" keep the raw path, so an unreachable literal is never reinterpreted. Grep and read use the same probe: the explicit selector branch keeps the literal path when existence is uncertain, and only a definitive ENOENT/ENOTDIR lets structured archive/sqlite/pdf dispatch take over. Added regressions covering probeLiteralPathExists exists/missing/dangling- symlink cases and splitPathAndSelPreferringLiteral over a dangling symlink. --- packages/coding-agent/src/tools/grep.ts | 15 +++++--- packages/coding-agent/src/tools/path-utils.ts | 37 ++++++++++++------- packages/coding-agent/src/tools/read.ts | 4 +- .../tools/path-literal-colon-selector.test.ts | 32 +++++++++++++++- 4 files changed, 65 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index 76d90f84c..393078133 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -48,8 +48,8 @@ import { type LineRange, parseLineRanges, pathTargetsSsh, + probeLiteralPathExists, type ResolvedSearchTarget, - resolveExistingReadPath, resolveReadPath, resolveToolSearchScope, selectorLineRanges, @@ -171,13 +171,16 @@ async function parsePathSpecs( // Separate selector parameter makes `path` deterministic: first try the // exact local filesystem path (with read-path normalization), then let // archive/internal/URL resolution handle non-literal structured paths. - const localPath = /^[a-z][a-z0-9+.-]*:\/\//i.test(entry) - ? undefined - : await resolveExistingReadPath(entry, cwd); + const rawPathHasScheme = /^[a-z][a-z0-9+.-]*:\/\//i.test(entry); + const probe = rawPathHasScheme ? "missing" : await probeLiteralPathExists(entry, cwd); + // `"unknown"` covers EACCES/IO where we cannot confirm existence — treat + // it as a literal so a real file such as `test:1-2` under an unreadable + // parent is never silently reinterpreted as `test` + selector. + const literalMatch = probe !== "missing"; specs.push({ original: entry, - clean: localPath ?? entry, - literalFilesystemMatch: localPath !== undefined, + clean: literalMatch && !rawPathHasScheme ? resolveReadPath(entry, cwd) : entry, + literalFilesystemMatch: literalMatch, ranges: explicitRanges, }); continue; diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index 807abc060..d5fd2f4de 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -315,25 +315,35 @@ export function splitPathAndSel(rawPath: string): { path: string; sel?: string } return { path: basePath, sel }; } -/** Resolve a read-tool path variant and return it only when it already exists on disk. */ -export async function resolveExistingReadPath(filePath: string, cwd: string): Promise { +/** + * Three-way probe for whether the exact filesystem entry named by `filePath` + * exists. `stat` (used earlier) failed for reasons other than "no such file" + * (dangling symlink, `EACCES` on a parent, transient I/O), and each of those + * silently reinterpreted a real literal path such as `test:1-2` as `test` + * plus selector `1-2` (issue #4618). `lstat` inspects the entry itself, so a + * dangling symlink is still detected as present; ambiguous errors resolve to + * `"unknown"` so callers keep the raw path instead of guessing. + */ +export async function probeLiteralPathExists(filePath: string, cwd: string): Promise<"exists" | "missing" | "unknown"> { const resolved = resolveReadPath(filePath, cwd); try { - await fs.promises.stat(resolved); - return resolved; + await fs.promises.lstat(resolved); + return "exists"; } catch (err) { - if (isEnoent(err) || isEnotdir(err)) return undefined; - return resolved; + if (isEnoent(err) || isEnotdir(err)) return "missing"; + return "unknown"; } } /** * Async sibling of {@link splitPathAndSel} that prefers a literal filesystem - * path over selector interpretation when the raw input exists on disk. - * Filenames whose tail matches the selector grammar (e.g. `test:1-2`, `log:raw`) - * are legal on POSIX; without this the strict splitter peels the tail and both - * `read` and `grep` refuse to open the real file (see issue #4618). Mirrors - * {@link parseSearchPathPreferringLiteral} for glob-shaped literal paths. + * path over selector interpretation. Filenames whose tail matches the selector + * grammar (e.g. `test:1-2`, `log:raw`) are legal on POSIX; without this the + * strict splitter peels the tail and both `read` and `grep` refuse to open the + * real file (issue #4618). The literal wins on a confirmed `lstat`, and also + * on `"unknown"` (`EACCES` on a parent, transient I/O), so an unreachable + * literal is never silently reinterpreted as `path + selector`. Only a + * definitive `ENOENT`/`ENOTDIR` falls back to the strict split. */ export async function splitPathAndSelPreferringLiteral( rawPath: string, @@ -341,9 +351,8 @@ export async function splitPathAndSelPreferringLiteral( ): Promise<{ path: string; sel?: string }> { const strict = splitPathAndSel(rawPath); if (strict.sel === undefined) return strict; - const resolved = await resolveExistingReadPath(rawPath, cwd); - if (resolved !== undefined) return { path: rawPath }; - return strict; + const probe = await probeLiteralPathExists(rawPath, cwd); + return probe === "missing" ? strict : { path: rawPath }; } /** diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 2412358b1..23d816bc1 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -99,7 +99,7 @@ import { type LineRange, parseLineRanges, pathTargetsSsh, - resolveExistingReadPath, + probeLiteralPathExists, resolveReadPath, splitDelimitedPathEntry, splitInternalUrlSel, @@ -2249,7 +2249,7 @@ export class ReadTool implements AgentTool { : { path: readPath, sel: explicitSelector }; const rawPathIsLiteral = explicitSelector !== undefined - ? readPath.includes(":") && (await resolveExistingReadPath(readPath, this.session.cwd)) !== undefined + ? readPath.includes(":") && (await probeLiteralPathExists(readPath, this.session.cwd)) !== "missing" : literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined; if (!rawPathIsLiteral) { diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts index 1792c1ca4..d6a01afcb 100644 --- a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -4,7 +4,11 @@ import * as os from "node:os"; import * as path from "node:path"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; -import { splitPathAndSel, splitPathAndSelPreferringLiteral } from "@oh-my-pi/pi-coding-agent/tools/path-utils"; +import { + probeLiteralPathExists, + splitPathAndSel, + splitPathAndSelPreferringLiteral, +} from "@oh-my-pi/pi-coding-agent/tools/path-utils"; import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; import { GrepTool } from "../../src/tools/grep"; @@ -80,6 +84,14 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); }); + it("keeps a literal dangling symlink intact (lstat exists even though stat fails)", async () => { + const literal = path.join(tmpDir, "test:1-2"); + await fs.symlink(path.join(tmpDir, "missing-target"), literal); + + expect(await probeLiteralPathExists(literal, tmpDir)).toBe("exists"); + expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); + }); + it("returns the strict split unchanged when there is no selector tail", async () => { expect(await splitPathAndSelPreferringLiteral("plain.txt", tmpDir)).toEqual({ path: "plain.txt", @@ -87,6 +99,24 @@ describe("literal colon filename resolution (issue #4618)", () => { }); }); + describe("probeLiteralPathExists", () => { + it('returns "missing" for a path that clearly does not exist', async () => { + expect(await probeLiteralPathExists(path.join(tmpDir, "never-here:1-2"), tmpDir)).toBe("missing"); + }); + + it('returns "exists" for a regular file', async () => { + const literal = path.join(tmpDir, "regular:1-2"); + await Bun.write(literal, "hi\n"); + expect(await probeLiteralPathExists(literal, tmpDir)).toBe("exists"); + }); + + it('returns "exists" for a dangling symlink', async () => { + const literal = path.join(tmpDir, "dangling:1-2"); + await fs.symlink(path.join(tmpDir, "nowhere"), literal); + expect(await probeLiteralPathExists(literal, tmpDir)).toBe("exists"); + }); + }); + describe("read tool", () => { it("reads a literal file whose name ends in a selector-shaped suffix", async () => { const literal = "test:1-2"; From b0a84847b7cd6f81329be83afb1f20b3f5603b77 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 3 Jul 2026 13:46:10 +0000 Subject: [PATCH 25/91] fix(agent): emitted handoff session switch hook Fixes #4434 --- packages/coding-agent/CHANGELOG.md | 4 + .../src/extensibility/shared-events.ts | 4 +- .../coding-agent/src/session/agent-session.ts | 17 ++++ .../test/agent-session-handoff.test.ts | 84 +++++++++++++++++++ 4 files changed, 107 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 860266c53..18b8c7e23 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting `session_switch` with `reason: "handoff"` before replacing the outgoing session ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). + ## [16.3.4] - 2026-07-03 ### Fixed diff --git a/packages/coding-agent/src/extensibility/shared-events.ts b/packages/coding-agent/src/extensibility/shared-events.ts index 7eb122e9e..8b8a4809e 100644 --- a/packages/coding-agent/src/extensibility/shared-events.ts +++ b/packages/coding-agent/src/extensibility/shared-events.ts @@ -33,7 +33,7 @@ export interface SessionStartEvent { export interface SessionBeforeSwitchEvent { type: "session_before_switch"; /** Reason for the switch */ - reason: "new" | "resume" | "fork"; + reason: "new" | "resume" | "fork" | "handoff"; /** Session file we're switching to (only for "resume") */ targetSessionFile?: string; } @@ -42,7 +42,7 @@ export interface SessionBeforeSwitchEvent { export interface SessionSwitchEvent { type: "session_switch"; /** Reason for the switch */ - reason: "new" | "resume" | "fork"; + reason: "new" | "resume" | "fork" | "handoff"; /** Session file we came from */ previousSessionFile: string | undefined; } diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index f271f21bf..51608ee6c 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -9901,6 +9901,16 @@ export class AgentSession { // Start a new session const previousSessionFile = this.sessionFile; + if (this.#extensionRunner?.hasHandlers("session_before_switch")) { + const result = (await this.#extensionRunner.emit({ + type: "session_before_switch", + reason: "handoff", + })) as SessionBeforeSwitchResult | undefined; + + if (result?.cancel) { + return undefined; + } + } await this.sessionManager.flush(); this.#cancelOwnAsyncJobs(); await this.sessionManager.newSession(previousSessionFile ? { parentSession: previousSessionFile } : undefined); @@ -9956,6 +9966,13 @@ export class AgentSession { this.agent.replaceMessages(sessionContext.messages); this.#resetAllAdvisorRuntimes(); this.#syncTodoPhasesFromBranch(); + if (this.#extensionRunner) { + await this.#extensionRunner.emit({ + type: "session_switch", + reason: "handoff", + previousSessionFile, + }); + } return { document: handoffText, savedPath }; } catch (error) { diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index 516a4f906..cce5bbd9f 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -158,6 +158,90 @@ describe("AgentSession handoff", () => { expect(sessionManager.getEntries().filter(entry => entry.type === "compaction")).toHaveLength(0); }); + it("emits handoff lifecycle hooks on the outgoing and replacement sessions", async () => { + const extensionsResult = await loadExtensions([], tempDir.path()); + const extensionRunner = new ExtensionRunner( + extensionsResult.extensions, + extensionsResult.runtime, + tempDir.path(), + sessionManager, + modelRegistry, + ); + const observedEvents: Array<{ + type: "session_before_switch" | "session_switch"; + reason: string; + previousSessionFile: string | undefined; + activeSessionFile: string | undefined; + messageCount: number; + handoffEntryCount: number; + }> = []; + vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch"); + const emit = extensionRunner.emit.bind(extensionRunner); + vi.spyOn(extensionRunner, "emit").mockImplementation(event => { + if (event.type === "session_before_switch" || event.type === "session_switch") { + observedEvents.push({ + type: event.type, + reason: event.reason, + previousSessionFile: event.type === "session_switch" ? event.previousSessionFile : undefined, + activeSessionFile: session.sessionFile, + messageCount: sessionManager.getBranch().filter(entry => entry.type === "message").length, + handoffEntryCount: sessionManager + .getBranch() + .filter(entry => entry.type === "custom_message" && entry.customType === "handoff").length, + }); + } + return emit(event); + }); + + await session.dispose(); + session = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }), + sessionManager, + settings: Settings.isolated({ + "compaction.enabled": true, + "compaction.autoContinue": false, + }), + modelRegistry, + extensionRunner, + obfuscator, + }); + const previousSessionFile = session.sessionFile; + const generateHandoffSpy = vi + .spyOn(compactionModule, "generateHandoffFromContext") + .mockResolvedValue("## Goal\nContinue from here"); + + await session.handoff(); + + const nextSessionFile = session.sessionFile; + expect(generateHandoffSpy).toHaveBeenCalledTimes(1); + expect(nextSessionFile).not.toBe(previousSessionFile); + expect(observedEvents).toEqual([ + { + type: "session_before_switch", + reason: "handoff", + previousSessionFile: undefined, + activeSessionFile: previousSessionFile, + messageCount: 2, + handoffEntryCount: 0, + }, + { + type: "session_switch", + reason: "handoff", + previousSessionFile, + activeSessionFile: nextSessionFile, + messageCount: 0, + handoffEntryCount: 1, + }, + ]); + }); + it("runs handoff generation through the configured side stream function", async () => { const handoffText = "## Goal\nContinue via side stream"; let sideStreamCalls = 0; From b2f7238a03357f2d928784a8619edb01cd2f86f9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 19:21:08 +0000 Subject: [PATCH 26/91] fix(coding-agent): allowed mixed fallback chains - Allowed implicit default fallback resolution when other role fallback chains are configured. - Covered the mixed-role first-run fallback case. Fixes #4533 --- packages/coding-agent/src/session/agent-session.ts | 11 ++--------- .../test/agent-session-retry-fallback.test.ts | 6 ++++-- 2 files changed, 6 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index e44bd7d81..b57ccbca0 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -13122,10 +13122,7 @@ export class AgentSession { if ( Array.isArray(defaultChain) && defaultChain.length > 0 && - this.#getRetryFallbackPrimarySelector("default") === undefined && - Object.entries(chains).every( - ([role, chain]) => role === "default" || !Array.isArray(chain) || chain.length === 0, - ) + this.#getRetryFallbackPrimarySelector("default") === undefined ) { return "default"; } @@ -13155,11 +13152,7 @@ export class AgentSession { if ( Array.isArray(defaultChain) && defaultChain.length > 0 && - this.#getRetryFallbackPrimarySelector("default") === undefined && - Object.entries(chains).every( - ([chainRole, roleChain]) => - chainRole === "default" || !Array.isArray(roleChain) || roleChain.length === 0, - ) + this.#getRetryFallbackPrimarySelector("default") === undefined ) { const seen = new Set([parsedCurrent.raw]); chain = [parsedCurrent]; diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 1adeda0ec..345d42df3 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -219,10 +219,11 @@ describe("AgentSession retry fallback", () => { ]); }); - it("uses the active initial model as the default fallback primary when the default role is unset", async () => { + it("uses the active initial model as the default fallback primary when other role fallback chains are configured", async () => { const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); const fallbackModel = getBundledModel("openai", "gpt-4o-mini"); - if (!primaryModel || !fallbackModel) { + const otherRoleFallbackModel = getBundledModel("openai", "gpt-4o"); + if (!primaryModel || !fallbackModel || !otherRoleFallbackModel) { throw new Error("Expected bundled test models to exist"); } @@ -235,6 +236,7 @@ describe("AgentSession retry fallback", () => { "retry.maxRetries": 1, "retry.fallbackChains": { default: [`${fallbackModel.provider}/${fallbackModel.id}`], + smol: [`${otherRoleFallbackModel.provider}/${otherRoleFallbackModel.id}`], }, }); From ea7c911ff2aa2e1be5a94d557f632ff2935a0b4e Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 19:22:56 +0000 Subject: [PATCH 27/91] fix(tui): rejected stale mid-prompt skill autocomplete on non-matching tokens MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The staleness guard previously accepted any clean trailing slash token, so the 100 ms debounce window let a bare-`/` skill popup rewrite a subsequently typed `/tmp` to `/skill:…`. Score the live token against the selected skill's value and reject the branch when it would no longer surface — Tab then falls through to file completion. Fixes #4619 --- packages/tui/src/autocomplete.ts | 2 +- packages/tui/src/components/editor.ts | 12 +++++++++++- .../tui/test/editor-autocomplete-actions.test.ts | 13 +++++++++++++ 3 files changed, 25 insertions(+), 2 deletions(-) diff --git a/packages/tui/src/autocomplete.ts b/packages/tui/src/autocomplete.ts index 63fad6c5a..45af4aa0d 100644 --- a/packages/tui/src/autocomplete.ts +++ b/packages/tui/src/autocomplete.ts @@ -276,7 +276,7 @@ function commandMatchesNameOrAlias(cmd: CommandEntry, commandName: string): bool return getCommandAliases(cmd).includes(commandName); } -function scoreCommandTextMatch(lowerPrefix: string, lowerTarget: string): number { +export function scoreCommandTextMatch(lowerPrefix: string, lowerTarget: string): number { if (lowerPrefix.length === 0) return 1; if (lowerPrefix === lowerTarget) return 1000; // Flat score for every prefix match so same-prefix commands keep registry diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index a1a354351..dfdb21166 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -3,6 +3,7 @@ import { type AutocompleteProvider, findLeadingSlashCommandStart, findTrailingSlashCommandStart, + scoreCommandTextMatch, } from "../autocomplete"; import { BracketedPasteHandler, decodeReencodedPasteControls } from "../bracketed-paste"; import { getKeybindings, type KeybindingsManager } from "../keybindings"; @@ -2902,8 +2903,17 @@ export class Editor implements Component, Focusable { const currentTrailingStart = findTrailingSlashCommandStart(currentTextBeforeCursor); if (currentTrailingStart !== null) { const token = currentTextBeforeCursor.slice(currentTrailingStart); - if (!token.includes(" ") && !token.slice(1).includes("/")) return true; + if (!token.includes(" ") && !token.slice(1).includes("/")) { + // Guard the timing window where the popup was built for an earlier + // query (e.g. bare `/`) and the user typed further characters before + // the 100 ms debounced refresh fired: accept the stale skill only + // when the current query would still surface it. `tmp` after a bare + // slash therefore falls through to file completion instead of + // rewriting the user's `/tmp` to `/skill:…`. + if (scoreCommandTextMatch(token.slice(1).toLowerCase(), item.value.toLowerCase()) > 0) return true; + } } + return false; } if (findLeadingSlashCommandStart(this.#autocompletePrefix) !== null) { diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index 2b926c410..b06029c80 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -188,6 +188,19 @@ describe("Editor Enter handler sync slash completion", () => { expect(editor.isShowingAutocomplete()).toBe(false); }); + it("does not apply a stale mid-prompt skill suggestion when the live token stops matching", async () => { + const editor = createSkillEditor(); + + await openMidPromptSkillAutocomplete(editor, "see "); + // Race the 100 ms debounce: type a non-skill token before the popup refreshes. + editor.handleInput("tmp"); + editor.handleInput("\t"); + + // The stale `skill:security-scan` popup must not rewrite `/tmp` to `/skill:…`. + expect(editor.getText()).toBe("see /tmp"); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + it("opens mid-prompt skill autocomplete and inserts the skill token without wiping the draft on Tab", async () => { const editor = new Editor(defaultEditorTheme); editor.setAutocompleteProvider( From a2e055fa9c30e905c552b978fdce5d3b3b23c95e Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 19:24:05 +0000 Subject: [PATCH 28/91] fix(tools): closed two open literal-wins gaps flagged by codex MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two Codex bot findings from earlier PR reviews were still open. 1. local:// URL selector shadow (read.ts): the local:// branch resolved `local://foo:1-2` and rewrote readPath to `${localFile.path}:${sel}`, then let splitPathAndSelPreferringLiteral run on the synthesized string. A sibling literal `${localFile.path}:${sel}` file would win over the intended URL selector semantics. The branch now promotes the URL selector into the explicit-selector state and sets readPath = localFile.path, so downstream literal-preferring routing never re-splits the concatenation. 2. Delimited expansion before literal probe (path-utils.ts): grep called expandDelimitedPathEntries before parsePathSpecs, and splitDelimitedPathEntry only checked whether the peeled base of the entry resolved. A real POSIX file whose name contained a delimiter plus a selector-shaped tail (a;b:1-2) got split into ["a", "b:1-2"] and never reached the literal-preferring probe. splitDelimitedPathEntry now short-circuits on probeLiteralPathExists — "missing" is the only outcome that lets delimiter expansion run. Added regressions: `read local://notes.md:1-2` still slices the base file when a sibling `notes.md:1-2` literal exists, and grep searches a real `a;b:1-2` file without semicolon-splitting. --- packages/coding-agent/src/tools/path-utils.ts | 7 ++++- packages/coding-agent/src/tools/read.ts | 18 ++++++++---- .../test/tools/grep-internal-urls.test.ts | 29 +++++++++++++++++++ .../tools/path-literal-colon-selector.test.ts | 18 ++++++++++++ 4 files changed, 65 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index d5fd2f4de..cc5b787c6 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -709,7 +709,12 @@ export async function splitDelimitedPathEntry( const normalizedEntry = normalizePathLikeInput(entry); if (!hasTopLevelPathDelimiter(normalizedEntry)) return null; if (isInternalUrlPath(normalizedEntry)) return null; - + // A real POSIX file may contain the delimiter and a selector-shaped tail + // (`a;b:1-2`, `a b:1-2`). Preserve the raw entry whenever the full literal + // resolves — or is only ambiguous — so downstream literal-preferring + // splitters see it before delimiter expansion peels or splits (issue #4618 + // reviewer feedback: delimited expansion ran before the literal check). + if ((await probeLiteralPathExists(normalizedEntry, cwd)) !== "missing") return null; const splitter = options.splitter ?? parseSearchPath; const peeledEntry = splitPathAndSel(normalizedEntry).path; if (!hasGlobPathChars(peeledEntry) && (await delimitedPathPartResolves(normalizedEntry, cwd, splitter))) { diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 23d816bc1..14edfad2b 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -2118,8 +2118,8 @@ export class ReadTool implements AgentTool { _toolContext?: AgentToolContext, ): Promise> { let { path: readPath } = params; - const explicitSelector = params.selector?.trim(); - const explicitParsedSelector = explicitSelector === undefined ? undefined : parseSel(explicitSelector); + let explicitSelector = params.selector?.trim(); + let explicitParsedSelector = explicitSelector === undefined ? undefined : parseSel(explicitSelector); if ( params.selector !== undefined && (explicitSelector === undefined || explicitSelector.length === 0 || explicitParsedSelector?.kind === "none") @@ -2221,10 +2221,16 @@ export class ReadTool implements AgentTool { skills: this.session.skills, }); if (localFile) { - readPath = - explicitSelector !== undefined || internalTarget.sel === undefined - ? localFile.path - : `${localFile.path}:${internalTarget.sel}`; + readPath = localFile.path; + // Promote the URL-embedded selector into the explicit-selector state so + // downstream literal-preferring routing does NOT re-split the synthesized + // `${localFile.path}:${sel}` string — a sibling literal file at that name + // would otherwise shadow the intended local:// URL selector semantics + // (issue #4618 reviewer feedback on c493d12). + if (explicitSelector === undefined && internalTarget.sel !== undefined) { + explicitSelector = internalTarget.sel; + explicitParsedSelector = parsed; + } } else { return this.#handleInternalUrl(internalTarget.path, parsed, signal); } diff --git a/packages/coding-agent/test/tools/grep-internal-urls.test.ts b/packages/coding-agent/test/tools/grep-internal-urls.test.ts index 417402f7c..b0d2e48ae 100644 --- a/packages/coding-agent/test/tools/grep-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/grep-internal-urls.test.ts @@ -450,6 +450,35 @@ describe("GrepTool internal URL resolution", () => { expect(text).toMatch(/^\*\d+:.*needle/m); }); + it("read local://: honors URL selector even when a sibling literal `:` file exists (issue #4618)", async () => { + const localRoot = path.join(artifactsDir, "local"); + await fs.mkdir(localRoot, { recursive: true }); + // Base file targeted by `local://notes.md`; selector should slice this one. + await Bun.write( + path.join(localRoot, "notes.md"), + `${Array.from({ length: 10 }, (_, i) => `url-target line ${i + 1}`).join("\n")}\n`, + ); + // Sibling literal `notes.md:1-2` under the same local root — must NOT + // shadow the URL selector semantics of `local://notes.md:1-2`. + await Bun.write(path.join(localRoot, "notes.md:1-2"), "sibling literal shadow\n"); + + LocalProtocolHandler.setOverride({ getArtifactsDir: () => artifactsDir, getSessionId: () => "session" }); + + const session = createSession({ hasEditTool: true }); + session.settings.set("read.summarize.enabled", false); + const result = await new ReadTool(session).execute("test-read-local-url-selector", { + path: "local://notes.md:1-2", + }); + + const text = getResultText(result); + // The base file was targeted (URL selector semantics preserved), not the + // sibling literal. Content check is enough — the read tool's context + // expansion around the requested range is unrelated to the shadow bug. + expect(text).toContain("url-target line 1"); + expect(text).toContain("url-target line 2"); + expect(text).not.toContain("sibling literal shadow"); + }); + it("keeps hashlines on mutable files when mixed with immutable artifact:// inputs", async () => { const content = "alpha line\nbeta needle line\ngamma line\n"; await Bun.write(path.join(artifactsDir, "11.bash.log"), content); diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts index d6a01afcb..e2a7d2544 100644 --- a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -289,6 +289,24 @@ describe("literal colon filename resolution (issue #4618)", () => { expect(output).toContain("escaped literal needle"); }); + it("searches a literal file whose name contains a semicolon and selector-shaped tail (`a;b:1-2`)", async () => { + // Semicolon is the delimited-path separator; without a raw-literal + // probe in `splitDelimitedPathEntry`, expandDelimitedPathEntries would + // split `a;b:1-2` into `["a", "b:1-2"]` before grep saw the literal file. + const literal = path.join(tmpDir, "a;b:1-2"); + await Bun.write(literal, "delimited literal needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-literal-semicolon-selector", { + pattern: "needle", + path: literal, + }); + const output = getText(result); + + expect(output).toContain("delimited literal needle"); + expect(output).not.toMatch(/not found/i); + }); + it("searches a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { // The base archive exists too; grep must not rematerialize the raw // literal path as archive `data.zip` plus phantom member `1-2`. From 241beb9b3d261196d5e10cb71508bd180d78c732 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 20:53:06 +0000 Subject: [PATCH 29/91] fix(agent): supported chat completions remote compaction - Sent OpenAI-compatible chat messages when compaction.remoteEndpoint targets /chat/completions while preserving the existing custom summarizer payload elsewhere. - Added regressions for direct wire formatting and end-to-end openai-completions compaction against a configured chat endpoint. Fixes #4630 --- docs/compaction.md | 6 +- packages/agent/CHANGELOG.md | 4 + packages/agent/src/compaction/compaction.ts | 38 +++--- packages/agent/src/compaction/openai.ts | 70 ++++++++++- packages/agent/test/remote-compaction.test.ts | 115 ++++++++++++++++++ 5 files changed, 211 insertions(+), 22 deletions(-) diff --git a/docs/compaction.md b/docs/compaction.md index 2112875eb..186d53054 100644 --- a/docs/compaction.md +++ b/docs/compaction.md @@ -240,9 +240,9 @@ Prompt selection: Remote summarization modes: -- If `compaction.remoteEndpoint` is set and remote compaction is enabled, local summary generation POSTs: - - `{ systemPrompt, prompt }` -- Expects JSON containing at least `{ summary }`. +- If `compaction.remoteEndpoint` is set and remote compaction is enabled, local summary generation POSTs one of two wire formats: + - custom omp summarizer endpoints receive `{ systemPrompt, prompt }` and must return JSON containing at least `{ summary }`. + - OpenAI-compatible endpoints whose path ends in `/chat/completions` receive `{ model, messages, stream: false }`, where `messages` contains one system prompt and one user prompt. The summary is read from `choices[0].message.content`, which lets self-hosted servers such as llama.cpp and vLLM act as remote compactors without a separate summarizer shim. - For OpenAI/OpenAI Codex models, compaction first tries the provider-native `/responses/compact` endpoint when remote compaction is enabled. It preserves provider replacement history in `preserveData.openaiRemoteCompaction` and falls back to local summarization if that native request fails. ### Handoff generation diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 1de3328af..a4d69bd2e 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed generic remote compaction against OpenAI-compatible `/chat/completions` endpoints (for example llama.cpp `openai-completions`) by sending chat messages instead of the custom `{ systemPrompt, prompt }` summarizer payload. ([#4630](https://github.com/can1357/oh-my-pi/issues/4630)) + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 9f502bac5..16134d2c1 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -813,14 +813,17 @@ export async function generateSummary( ]; if (options?.remoteEndpoint) { - const remote = await requestRemoteCompaction( - options.remoteEndpoint, - { - systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, - prompt: promptText, - }, - signal, - { fetch: options.fetch }, + const endpoint = options.remoteEndpoint; + const remote = await withAuth( + apiKey, + key => + requestRemoteCompaction( + endpoint, + { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, prompt: promptText }, + signal, + { fetch: options.fetch, model, apiKey: key }, + ), + { signal, missingKeyMessage: "Remote compaction credentials unavailable" }, ); return remote.summary; } @@ -1001,14 +1004,17 @@ async function generateShortSummary( promptText += SHORT_SUMMARY_PROMPT; if (options?.remoteEndpoint) { - const remote = await requestRemoteCompaction( - options.remoteEndpoint, - { - systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, - prompt: promptText, - }, - signal, - { fetch: options?.fetch }, + const endpoint = options.remoteEndpoint; + const remote = await withAuth( + apiKey, + key => + requestRemoteCompaction( + endpoint, + { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, prompt: promptText }, + signal, + { fetch: options?.fetch, model, apiKey: key }, + ), + { signal, missingKeyMessage: "Remote compaction credentials unavailable" }, ); return remote.summary; } diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index 5a65d7507..554ed3526 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -547,16 +547,55 @@ export async function requestOpenAiRemoteCompaction( return { provider: model.provider, replacementHistory, compactionItem }; } +/** + * Generic remote-compaction POST. Two wire shapes are auto-selected by + * endpoint suffix so a single `compaction.remoteEndpoint` setting can point at + * either a purpose-built omp summarizer (`{systemPrompt, prompt}` → `{summary}`) + * or any OpenAI-compatible chat-completions server (`/chat/completions`, + * `/v1/chat/completions`, …) as reported for llama.cpp / vLLM / etc. in + * issue #4630: without this, the omp payload was rejected with + * HTTP 400 `"'messages' is required"`, compaction silently fell back to + * local summarization, and context grew unbounded. + * + * When `context.model` is provided the chat-completions body is tagged with + * that model id (llama.cpp requires the field) and `context.apiKey` is + * forwarded as `Authorization: Bearer`. Callers wrap this in `withAuth` so + * 401s force-refresh through the standard credential rotation policy. + */ export async function requestRemoteCompaction( endpoint: string, request: RemoteCompactionRequest, signal?: AbortSignal, - opts?: { fetch?: FetchImpl; timeoutMs?: number }, + opts?: { fetch?: FetchImpl; timeoutMs?: number; model?: Model; apiKey?: string }, ): Promise { + let endpointPath = endpoint; + try { + endpointPath = new URL(endpoint).pathname; + } catch { + // Keep the raw endpoint for relative/custom fetch implementations. + } + const isChatCompletions = /\/chat\/completions\/?$/.test(endpointPath); + const headers: Record = { "content-type": "application/json" }; + if (isChatCompletions) { + if (opts?.apiKey) headers.Authorization = `Bearer ${opts.apiKey}`; + if (opts?.model?.headers) Object.assign(headers, opts.model.headers); + } + + const body: Record = isChatCompletions + ? { + model: opts?.model?.id, + messages: [ + { role: "system", content: request.systemPrompt }, + { role: "user", content: request.prompt }, + ], + stream: false, + } + : { systemPrompt: request.systemPrompt, prompt: request.prompt }; + const response = await (opts?.fetch ?? fetch)(endpoint, { method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify(request), + headers, + body: JSON.stringify(body), signal: withRequestTimeout(signal, opts?.timeoutMs ?? REMOTE_COMPACTION_TIMEOUT_MS), }); @@ -577,6 +616,31 @@ export async function requestRemoteCompaction( ); } + if (isChatCompletions) { + type ChatCompletionsResponse = { + choices?: Array<{ + message?: { + content?: string | Array<{ type?: string; text?: string }> | null; + }; + }>; + }; + const data = (await response.json()) as ChatCompletionsResponse | undefined; + const choice = data?.choices?.[0]?.message?.content; + let summary: string | undefined; + if (typeof choice === "string") { + summary = choice; + } else if (Array.isArray(choice)) { + summary = choice + .filter((part): part is { type?: string; text: string } => typeof part?.text === "string") + .map(part => part.text) + .join(""); + } + if (typeof summary !== "string" || summary.length === 0) { + throw new Error("Remote compaction response missing choices[0].message.content"); + } + return { summary }; + } + const data = (await response.json()) as RemoteCompactionResponse | undefined; if (!data || typeof data.summary !== "string") { throw new Error("Remote compaction response missing summary"); diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index d88104b18..65efee996 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -13,6 +13,7 @@ import { getCompactionV2PreserveData, requestCompactionV2Streaming, requestOpenAiRemoteCompaction, + requestRemoteCompaction, shouldUseCompactionV2Streaming, shouldUseOpenAiRemoteCompaction, } from "@oh-my-pi/pi-agent-core/compaction/openai"; @@ -540,6 +541,74 @@ describe("requestOpenAiRemoteCompaction timeout", () => { }); }); +describe("requestRemoteCompaction wire formats", () => { + test("uses OpenAI chat completions format for /chat/completions endpoints", async () => { + const model = buildModel({ + id: "Jackrong/Qwopus3.6-35B-A3B-Coder", + name: "Qwopus 3.6 35B-A3B Coder", + api: "openai-completions", + provider: "local-llama", + baseUrl: "http://127.0.0.1:8001/v1", + headers: { "x-local-llama": "1" }, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131072, + maxTokens: 4096, + }); + let sentBody: unknown; + const fetchMock: FetchImpl = async (_input, init) => { + if (typeof init?.body !== "string") throw new Error("missing remote compaction request body"); + sentBody = JSON.parse(init.body) as unknown; + const headers = new Headers(init.headers); + expect(headers.get("authorization")).toBe("Bearer local-key"); + expect(headers.get("x-local-llama")).toBe("1"); + return new Response(JSON.stringify({ choices: [{ message: { content: "remote summary" } }] }), { + headers: { "content-type": "application/json" }, + }); + }; + + const result = await requestRemoteCompaction( + "http://127.0.0.1:8001/v1/chat/completions", + { systemPrompt: "summarize", prompt: "hello" }, + undefined, + { fetch: fetchMock, model, apiKey: "local-key" }, + ); + + expect(result).toEqual({ summary: "remote summary" }); + expect(sentBody).toEqual({ + model: "Jackrong/Qwopus3.6-35B-A3B-Coder", + messages: [ + { role: "system", content: "summarize" }, + { role: "user", content: "hello" }, + ], + stream: false, + }); + }); + + test("keeps the generic omp summarizer format for other endpoints", async () => { + let sentBody: unknown; + const fetchMock: FetchImpl = async (_input, init) => { + if (typeof init?.body !== "string") throw new Error("missing remote compaction request body"); + sentBody = JSON.parse(init.body) as unknown; + expect(new Headers(init.headers).get("authorization")).toBeNull(); + return new Response(JSON.stringify({ summary: "generic summary", shortSummary: "generic" }), { + headers: { "content-type": "application/json" }, + }); + }; + + const result = await requestRemoteCompaction( + "https://compaction.example.test/summarize", + { systemPrompt: "summarize", prompt: "hello" }, + undefined, + { fetch: fetchMock, apiKey: "unused-for-generic" }, + ); + + expect(result).toEqual({ summary: "generic summary", shortSummary: "generic" }); + expect(sentBody).toEqual({ systemPrompt: "summarize", prompt: "hello" }); + }); +}); + describe("compact() remote compaction failure handling", () => { afterEach(() => { vi.restoreAllMocks(); @@ -771,6 +840,52 @@ describe("compact() remote compaction failure handling", () => { expect(completeSpy).not.toHaveBeenCalled(); }); + test("uses configured chat completions endpoints for openai-completions remote compaction", async () => { + const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local fallback")); + const preparation = makePreparation(); + preparation.settings = { + ...preparation.settings, + remoteEndpoint: "http://127.0.0.1:8001/v1/chat/completions", + remoteStreamingV2Enabled: false, + }; + const model = buildModel({ + id: "Jackrong/Qwopus3.6-35B-A3B-Coder", + name: "Qwopus 3.6 35B-A3B Coder", + api: "openai-completions", + provider: "local-llama", + baseUrl: "http://127.0.0.1:8001/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131072, + maxTokens: 4096, + }); + const requestBodies: unknown[] = []; + const fetchMock: FetchImpl = async (_input, init) => { + if (typeof init?.body !== "string") throw new Error("missing remote compaction request body"); + requestBodies.push(JSON.parse(init.body) as unknown); + expect(new Headers(init.headers).get("authorization")).toBe("Bearer local-key"); + const summary = requestBodies.length === 1 ? "remote history summary" : "remote short summary"; + return new Response(JSON.stringify({ choices: [{ message: { content: summary } }] }), { + headers: { "content-type": "application/json" }, + }); + }; + + const result = await compact(preparation, model, "local-key", undefined, undefined, { + fetch: fetchMock, + }); + + expect(result.summary).toContain("remote history summary"); + expect(result.shortSummary).toBe("remote short summary"); + expect(completeSpy).not.toHaveBeenCalled(); + expect(requestBodies).toHaveLength(2); + expect(requestBodies[0]).toMatchObject({ + model: "Jackrong/Qwopus3.6-35B-A3B-Coder", + messages: [{ role: "system" }, { role: "user", content: expect.stringContaining("long history") }], + stream: false, + }); + }); + test("remote compact server failure without abort still falls back to local summarization", async () => { const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local summary")); const fetchMock: FetchImpl = async () => From da6f0ebc381cbc1f4e5405dd72c14d058d4ec231 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 5 Jul 2026 21:12:08 +0000 Subject: [PATCH 30/91] fix(agent): used wire model id for chat compaction - Sent remoteCompaction.model or requestModelId in chat-completions remote compaction requests instead of the local catalog id. - Covered both direct requestRemoteCompaction formatting and end-to-end openai-completions compaction with wire model ids. Fixes #4630 --- packages/agent/src/compaction/openai.ts | 4 ++-- packages/agent/test/remote-compaction.test.ts | 11 +++++++---- 2 files changed, 9 insertions(+), 6 deletions(-) diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index 554ed3526..2695e4466 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -558,7 +558,7 @@ export async function requestOpenAiRemoteCompaction( * local summarization, and context grew unbounded. * * When `context.model` is provided the chat-completions body is tagged with - * that model id (llama.cpp requires the field) and `context.apiKey` is + * that model's wire id (llama.cpp requires the field) and `context.apiKey` is * forwarded as `Authorization: Bearer`. Callers wrap this in `withAuth` so * 401s force-refresh through the standard credential rotation policy. */ @@ -583,7 +583,7 @@ export async function requestRemoteCompaction( const body: Record = isChatCompletions ? { - model: opts?.model?.id, + model: opts?.model ? resolveOpenAiCompactModel(opts.model) : undefined, messages: [ { role: "system", content: request.systemPrompt }, { role: "user", content: request.prompt }, diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index 65efee996..c2cc50dd1 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -544,8 +544,10 @@ describe("requestOpenAiRemoteCompaction timeout", () => { describe("requestRemoteCompaction wire formats", () => { test("uses OpenAI chat completions format for /chat/completions endpoints", async () => { const model = buildModel({ - id: "Jackrong/Qwopus3.6-35B-A3B-Coder", + id: "catalog-selection-id", name: "Qwopus 3.6 35B-A3B Coder", + requestModelId: "provider-wire-id", + remoteCompaction: { model: "provider-compact-wire-id" }, api: "openai-completions", provider: "local-llama", baseUrl: "http://127.0.0.1:8001/v1", @@ -577,7 +579,7 @@ describe("requestRemoteCompaction wire formats", () => { expect(result).toEqual({ summary: "remote summary" }); expect(sentBody).toEqual({ - model: "Jackrong/Qwopus3.6-35B-A3B-Coder", + model: "provider-compact-wire-id", messages: [ { role: "system", content: "summarize" }, { role: "user", content: "hello" }, @@ -849,8 +851,9 @@ describe("compact() remote compaction failure handling", () => { remoteStreamingV2Enabled: false, }; const model = buildModel({ - id: "Jackrong/Qwopus3.6-35B-A3B-Coder", + id: "catalog-selection-id", name: "Qwopus 3.6 35B-A3B Coder", + requestModelId: "provider-wire-id", api: "openai-completions", provider: "local-llama", baseUrl: "http://127.0.0.1:8001/v1", @@ -880,7 +883,7 @@ describe("compact() remote compaction failure handling", () => { expect(completeSpy).not.toHaveBeenCalled(); expect(requestBodies).toHaveLength(2); expect(requestBodies[0]).toMatchObject({ - model: "Jackrong/Qwopus3.6-35B-A3B-Coder", + model: "provider-wire-id", messages: [{ role: "system" }, { role: "user", content: expect.stringContaining("long history") }], stream: false, }); From 862f821d76118c46bd44b4b765b9ad1f8b70d6e5 Mon Sep 17 00:00:00 2001 From: Dylan Bohlender Date: Fri, 3 Jul 2026 02:35:04 -0600 Subject: [PATCH 31/91] fix(open): report Windows opener failures via Start-Process exit codes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the #4420 opener hardening: absolute-path rundll32 fixes the stripped-PATH spawn throw, but rundll32 exits 0 unconditionally, so the delayed-failure telemetry added there can never observe a Windows launch failure. Replace it with %SystemRoot%-resolved PowerShell Start-Process via -EncodedCommand: - failures ShellExecute itself reports (missing target, no handler executable, access denied) surface as exit code 1 and reach the existing non-zero-exit logging (verified live on Windows 11: missing file exits 1; unregistered schemes exit 0 on any opener because the OS hands them to the app-picker — documented limitation); - the UTF-16LE/base64 payload keeps OAuth query strings opaque to cmd/PowerShell metacharacter parsing; embedded single quotes are doubled into a PS literal; - %SystemRoot% anchoring with a bare-name PATH fallback preserves the stripped-PATH resilience from #4420. Also pins the WSL-mount test's path.resolve against Windows dev hosts so the mocked linux platform stays deterministic. Refs #4418 --- packages/coding-agent/src/utils/open.ts | 46 +++++++--- packages/coding-agent/test/utils/open.test.ts | 83 ++++++++++++++++--- 2 files changed, 106 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/src/utils/open.ts b/packages/coding-agent/src/utils/open.ts index ca3390438..413e9ca2e 100644 --- a/packages/coding-agent/src/utils/open.ts +++ b/packages/coding-agent/src/utils/open.ts @@ -32,22 +32,48 @@ function getExistingWslLocalPath(urlOrPath: string): string | undefined { } /** - * Resolve the Windows `rundll32.exe` command used to hand a URL/path to the - * user's registered protocol handler. Anchoring to `%SystemRoot%\System32` - * (rather than relying on `rundll32` being on `PATH`) survives environments - * where the machine `PATH` no longer references `System32` — a common - * real-world misconfiguration where `System32\Wbem` / `WindowsPowerShell` / - * `OpenSSH` survive but `System32` itself is dropped. Bare `rundll32` on - * such boxes throws `Executable not found in $PATH: "rundll32"` from - * `Bun.spawn` before ShellExecute ever sees the URL. + * Resolve the Windows opener used to hand a URL/path to the user's registered + * protocol handler. PowerShell's `Start-Process` goes through ShellExecute + * like the previous `rundll32 url.dll,FileProtocolHandler`, with two + * advantages that make the delayed-failure telemetry in {@link openPath} + * actually observable on Windows: + * + * - `rundll32` exits 0 unconditionally, so no launch failure ever reaches the + * non-zero-exit logging below. `Start-Process` surfaces the failures + * ShellExecute itself reports — missing target file, no handler executable, + * access denied — as exit code 1 (verified live: a nonexistent file path + * exits 1; `$ErrorActionPreference='Stop'` additionally promotes any + * non-terminating error classes). Known limitation shared by every opener: + * an unregistered URL scheme exits 0 because Windows "handles" it by + * offering the app-picker. + * - `-EncodedCommand` carries the target as a UTF-16LE/base64 payload, so no + * cmd/PowerShell metacharacter parsing ever sees it (OAuth authorize URLs + * carry `&`); inside the decoded script the target is a single-quoted + * literal (no `$` expansion) with embedded quotes doubled. + * + * PowerShell is anchored to `%SystemRoot%\System32` for the same reason the + * previous revision anchored `rundll32`: machine PATHs that dropped + * `System32` are a real-world occurrence, and bare names throw + * `Executable not found in $PATH` from `Bun.spawn`. A bare-name fallback + * remains for exotic SystemRoot layouts. */ function windowsOpenerCommand(target: string): string[] { const systemRoot = process.env.SystemRoot?.trim() || process.env.SYSTEMROOT?.trim() || "C:\\Windows"; // `path.win32` (not the platform-adaptive `path.join`) keeps Windows path // separators when tests run under a POSIX host and matches Windows call // conventions on the real target. - const rundll32 = path.win32.join(systemRoot, "System32", "rundll32.exe"); - return [rundll32, "url.dll,FileProtocolHandler", target]; + const absolute = path.win32.join(systemRoot, "System32", "WindowsPowerShell", "v1.0", "powershell.exe"); + const powershell = fs.existsSync(absolute) ? absolute : "powershell.exe"; + const script = `$ErrorActionPreference='Stop';Start-Process '${target.replaceAll("'", "''")}'`; + return [ + powershell, + "-NoProfile", + "-NonInteractive", + "-WindowStyle", + "Hidden", + "-EncodedCommand", + Buffer.from(script, "utf16le").toString("base64"), + ]; } /** Open a URL or file path in the default browser/application. Best-effort, never throws. */ export function openPath(urlOrPath: string): void { diff --git a/packages/coding-agent/test/utils/open.test.ts b/packages/coding-agent/test/utils/open.test.ts index 467c28f4e..11d28a46d 100644 --- a/packages/coding-agent/test/utils/open.test.ts +++ b/packages/coding-agent/test/utils/open.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; +import * as path from "node:path"; import { openPath } from "@oh-my-pi/pi-coding-agent/utils/open"; import * as piUtils from "@oh-my-pi/pi-utils"; import type { Subprocess } from "bun"; @@ -91,6 +92,12 @@ describe("openPath", () => { it("opens existing WSL mount files through wslview with a Windows path", () => { setPlatform("linux"); process.env.WSL_DISTRO_NAME = "Ubuntu"; + // Keep the mocked linux platform deterministic on a Windows dev host: + // the real path.resolve would rewrite /mnt/c/… against the drive root. + const realResolve = path.resolve; + vi.spyOn(path, "resolve").mockImplementation((...segments: string[]) => + segments.length === 1 && segments[0] === existingLinuxPath ? existingLinuxPath : realResolve(...segments), + ); vi.spyOn(piUtils, "$which").mockImplementation(command => (command === "wslview" ? "/usr/bin/wslview" : null)); vi.spyOn(fs, "existsSync").mockImplementation(candidate => candidate === existingLinuxPath); @@ -133,50 +140,100 @@ describe("openPath", () => { expect(spawnCalls.map(call => call.cmd)).toEqual([["xdg-open", existingLinuxPath]]); }); - it("resolves rundll32 through %SystemRoot% so a broken machine PATH cannot silence the opener", () => { + it("resolves PowerShell through %SystemRoot% so a broken machine PATH cannot silence the opener", () => { setPlatform("win32"); const originalSystemRoot = process.env.SystemRoot; process.env.SystemRoot = "D:\\CustomWindows"; + const powershellPath = "D:\\CustomWindows\\System32\\WindowsPowerShell\\v1.0\\powershell.exe"; + vi.spyOn(fs, "existsSync").mockImplementation(candidate => candidate === powershellPath); try { const spawnCalls: SpawnCall[] = []; spySpawn(spawnCalls); - openPath("https://mcp.linear.app/authorize?state=xyz&code_challenge_method=S256"); + const url = "https://mcp.linear.app/authorize?state=xyz&code_challenge_method=S256"; + openPath(url); expect(spawnCalls).toHaveLength(1); const [call] = spawnCalls; - // Absolute rundll32 path — bare `rundll32` was the whole bug on Windows - // boxes where the machine PATH no longer references System32. - expect(call?.cmd[0]).toBe("D:\\CustomWindows\\System32\\rundll32.exe"); - // Handler + URL forwarded verbatim as a single argv slot so `&` in the - // query string cannot be interpreted as a shell separator. - expect(call?.cmd.slice(1)).toEqual([ - "url.dll,FileProtocolHandler", - "https://mcp.linear.app/authorize?state=xyz&code_challenge_method=S256", + // Absolute PowerShell path — bare executable names were the whole bug + // on Windows boxes where the machine PATH no longer references + // System32. + expect(call?.cmd[0]).toBe(powershellPath); + expect(call?.cmd.slice(1, -1)).toEqual([ + "-NoProfile", + "-NonInteractive", + "-WindowStyle", + "Hidden", + "-EncodedCommand", ]); + // The target rides inside the UTF-16LE payload: no cmd/PowerShell + // metacharacter parsing ever sees the `&` in the query string, and the + // terminating error preference makes Start-Process failures exit 1 so + // openPath's non-zero-exit telemetry observes them (rundll32 always + // exited 0). + const decoded = Buffer.from(String(call?.cmd.at(-1)), "base64").toString("utf16le"); + expect(decoded).toBe(`$ErrorActionPreference='Stop';Start-Process '${url}'`); } finally { if (originalSystemRoot === undefined) delete process.env.SystemRoot; else process.env.SystemRoot = originalSystemRoot; } }); - it("falls back to C:\\Windows for rundll32 when SystemRoot is unset", () => { + it("doubles embedded single quotes so the target stays one PowerShell literal", () => { + setPlatform("win32"); + vi.spyOn(fs, "existsSync").mockReturnValue(true); + const spawnCalls: SpawnCall[] = []; + spySpawn(spawnCalls); + + openPath("C:\\Users\\o'brien\\report.html"); + + const decoded = Buffer.from(String(spawnCalls[0]?.cmd.at(-1)), "base64").toString("utf16le"); + expect(decoded).toBe("$ErrorActionPreference='Stop';Start-Process 'C:\\Users\\o''brien\\report.html'"); + }); + + it("falls back to C:\\Windows for PowerShell when SystemRoot is unset, and to the bare name when absent", () => { setPlatform("win32"); const originalSystemRoot = process.env.SystemRoot; const originalSystemRootLower = process.env.SYSTEMROOT; delete process.env.SystemRoot; delete process.env.SYSTEMROOT; try { + const defaultPath = "C:\\Windows\\System32\\WindowsPowerShell\\v1.0\\powershell.exe"; + const existsSpy = vi.spyOn(fs, "existsSync").mockImplementation(candidate => candidate === defaultPath); const spawnCalls: SpawnCall[] = []; spySpawn(spawnCalls); openPath("https://example.com"); + expect(spawnCalls[0]?.cmd[0]).toBe(defaultPath); - expect(spawnCalls).toHaveLength(1); - expect(spawnCalls[0]?.cmd[0]).toBe("C:\\Windows\\System32\\rundll32.exe"); + // Exotic layout: resolved path missing → bare name via PATH. + existsSpy.mockReturnValue(false); + openPath("https://example.com"); + expect(spawnCalls[1]?.cmd[0]).toBe("powershell.exe"); } finally { if (originalSystemRoot !== undefined) process.env.SystemRoot = originalSystemRoot; if (originalSystemRootLower !== undefined) process.env.SYSTEMROOT = originalSystemRootLower; } }); + + it("logs when the opener exits non-zero so Start-Process failures are diagnosable", async () => { + setPlatform("win32"); + vi.spyOn(fs, "existsSync").mockReturnValue(true); + const warnSpy = vi.spyOn(piUtils.logger, "warn").mockImplementation((() => {}) as never); + const failing = { + pid: 1, + exited: Promise.resolve(1), + kill: () => true, + } as unknown as Subprocess; + vi.spyOn(Bun, "spawn").mockImplementation(() => failing); + + openPath("https://example.com"); + await failing.exited; + await Promise.resolve(); + + expect(warnSpy).toHaveBeenCalledWith( + "External opener exited with non-zero status", + expect.objectContaining({ exitCode: 1 }), + ); + }); }); From 54f0a00dd25074e61cf9438a4439905d58f891de Mon Sep 17 00:00:00 2001 From: Dylan Bohlender Date: Sun, 5 Jul 2026 16:01:30 -0600 Subject: [PATCH 32/91] fix(oauth): copy-safe URL chunks and loopback-only launch URLs Resolves the two Codex P2s raised on #4420 that merged unaddressed: - wrapUrlRows indented every continuation chunk. A multi-row terminal selection includes the newline plus that indent; address bars strip newlines but preserve or percent-encode embedded spaces, so the reassembled URL was corrupted at every chunk boundary - silently, when the damage landed inside a query value. Chunk rows now carry zero leading bytes (label rows keep their indent), and the test reassembly helper concatenates chunks raw instead of stripping the indent that previously masked exactly this defect. - #launchUrlIfSafe advertised a localhost /launch copy target for flows whose redirectUri never returns to the loopback server. Its catch-comment assumed custom-scheme URIs are non-parseable, but new URL('vscode://gitlab.gitlab-workflow/authentication') parses fine and sailed through the pathname check. The guard now requires an http(s) loopback redirectUri (localhost / 127.0.0.1 / [::1]); custom schemes, non-loopback hosts, and unparseable URIs all suppress the launch URL. Regression tests cover the GitLab Duo vscode:// shape and a fixed non-loopback HTTPS redirect. Refs #4418 --- .../ai/src/registry/oauth/callback-server.ts | 26 ++++++--- .../test/callback-server-launch-route.test.ts | 57 +++++++++++++++++++ .../controllers/mcp-command-controller.ts | 19 ++++--- .../mcp-authorization-link.test.ts | 28 ++++----- 4 files changed, 101 insertions(+), 29 deletions(-) diff --git a/packages/ai/src/registry/oauth/callback-server.ts b/packages/ai/src/registry/oauth/callback-server.ts index afdbd1808..c3a45dc97 100644 --- a/packages/ai/src/registry/oauth/callback-server.ts +++ b/packages/ai/src/registry/oauth/callback-server.ts @@ -229,20 +229,32 @@ export abstract class OAuthCallbackFlow { /** * Build the `/launch` URL served by the callback server bound to `port`, or - * `undefined` when the configured `callbackPath` (or a `redirectUri` whose - * pathname resolves to {@link LAUNCH_PATH}) would collide with the launch - * route. Kept short (~30 chars) so UIs can advertise it as a + * `undefined` when it must not be advertised: + * - the configured `callbackPath` (or a `redirectUri` whose pathname + * resolves to {@link LAUNCH_PATH}) would collide with the launch route; + * - the flow's `redirectUri` never returns to this loopback server: fixed + * non-loopback hosts, or custom schemes like GitLab Duo's `vscode://` + * URI — which `new URL` parses without complaint, so a scheme/host check + * is required, not just the parse failure path. Advertising a localhost + * `/launch` target for such flows misrepresents the callback endpoint + * and hands remote users a URL that resolves nowhere. + * Kept short (~30 chars) so UIs can advertise it as a * viewport-truncation-safe copy target for the full authorization URL. */ #launchUrlIfSafe(port: number): string | undefined { if (this.callbackPath === LAUNCH_PATH) return undefined; if (this.redirectUri) { try { - if (new URL(this.redirectUri).pathname === LAUNCH_PATH) return undefined; + const parsed = new URL(this.redirectUri); + if (parsed.protocol !== "http:" && parsed.protocol !== "https:") return undefined; + if (parsed.hostname !== "localhost" && parsed.hostname !== "127.0.0.1" && parsed.hostname !== "[::1]") { + return undefined; + } + if (parsed.pathname === LAUNCH_PATH) return undefined; } catch { - // A non-parseable redirectUri (e.g. `vscode://...` handled elsewhere) - // can't collide with an HTTP `/launch` route — fall through and - // advertise the launch URL against the loopback server. + // A redirectUri even WHATWG URL cannot parse certainly does not + // return to this server — never advertise a launch URL for it. + return undefined; } } return `http://${this.callbackHostname}:${port}${LAUNCH_PATH}`; diff --git a/packages/ai/test/callback-server-launch-route.test.ts b/packages/ai/test/callback-server-launch-route.test.ts index 8f14cc59b..3d22b4205 100644 --- a/packages/ai/test/callback-server-launch-route.test.ts +++ b/packages/ai/test/callback-server-launch-route.test.ts @@ -182,4 +182,61 @@ describe("OAuthCallbackFlow /launch route", () => { abort.abort("test done"); await login; }); + + it("suppresses launchUrl for custom-scheme redirects that never return to the loopback server", async () => { + const abort = new AbortController(); + const authFired = Promise.withResolvers(); + const flow = new LaunchProbeFlow( + { + onAuth: info => { + authFired.resolve(info); + }, + signal: abort.signal, + }, + { + preferredPort: 0, + allowPortFallback: true, + // GitLab Duo shape: `new URL` parses this happily (pathname + // `/authentication`), so the guard must check scheme/host, not + // rely on a parse failure. A localhost /launch copy target for + // this flow would misrepresent the callback endpoint and point + // remote users at a URL that resolves nowhere. + redirectUri: "vscode://gitlab.gitlab-workflow/authentication", + }, + ); + const login = flow.login().catch(() => undefined) as Promise; + const info = await authFired.promise; + + expect(info.launchUrl).toBeUndefined(); + + abort.abort("test done"); + await login; + }); + + it("suppresses launchUrl for fixed non-loopback HTTP redirects", async () => { + const abort = new AbortController(); + const authFired = Promise.withResolvers(); + const flow = new LaunchProbeFlow( + { + onAuth: info => { + authFired.resolve(info); + }, + signal: abort.signal, + }, + { + preferredPort: 0, + allowPortFallback: true, + // The provider redirects to a hosted endpoint; this machine's + // callback server never sees the redirect, so no launch URL. + redirectUri: "https://auth.example.com/oauth/callback", + }, + ); + const login = flow.login().catch(() => undefined) as Promise; + const info = await authFired.promise; + + expect(info.launchUrl).toBeUndefined(); + + abort.abort("test done"); + await login; + }); }); diff --git a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts index 90040b394..5d0491a10 100644 --- a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts @@ -88,12 +88,14 @@ function raceAbortSignal(promise: Promise, signal: AbortSignal, createErro const MCP_AUTH_MIN_WRAP_WIDTH = 16; /** - * Wrap `url` into rows that each fit inside `width`, prefixed by a shared - * single-column indent so nested composition doesn't touch column 0. When the - * label + URL fit on one line, returns a single row; otherwise puts the label - * on its own row and slices the URL into fixed-width chunks. URL chunks are - * plain code points — browsers strip whitespace when pasted into the address - * bar, so a multi-row selection copies back to the intact URL. + * Wrap `url` into rows that each fit inside `width`. When the label + URL fit + * on one line, returns a single indented row; otherwise puts the label on its + * own indented row and slices the URL into fixed-width chunks that start at + * column 0. Continuation chunks carry ZERO leading bytes on purpose: a + * multi-row terminal selection includes the newline plus any leading indent, + * and while address bars strip newlines they preserve or percent-encode + * embedded spaces — an indent would corrupt the URL at every chunk boundary + * (silently, when the damage lands inside a query value). */ function wrapUrlRows(label: string, url: string, width: number): string[] { const indent = " "; @@ -103,10 +105,9 @@ function wrapUrlRows(label: string, url: string, width: number): string[] { if (inlineWidth <= effective) { return [`${indent}${theme.fg("muted", `${label} ${sanitized}`)}`]; } - const chunkWidth = Math.max(1, effective - indent.length); const rows: string[] = [`${indent}${theme.fg("muted", label)}`]; - for (let i = 0; i < sanitized.length; i += chunkWidth) { - rows.push(`${indent}${theme.fg("muted", sanitized.slice(i, i + chunkWidth))}`); + for (let i = 0; i < sanitized.length; i += effective) { + rows.push(theme.fg("muted", sanitized.slice(i, i + effective))); } return rows; } diff --git a/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts b/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts index 7c0bc7a04..019fbacae 100644 --- a/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts +++ b/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts @@ -24,10 +24,12 @@ const LONG_AUTH_URL = const LINEAR_AUTH_URL = `https://mcp.linear.app/authorize?response_type=code&client_id=abcdefghij0123456789ABCDEFGHIJ0123456789&redirect_uri=http%3A%2F%2Flocalhost%3A3000%2Fcallback&scope=read%20write%20mcp%3Aall&state=0123456789abcdef0123456789abcdef&code_challenge=5MlkJfN2GhX9uP0rQ7sT8vB1oCwDeFgHiJkLmNoPqRsTuVwXyZ&code_challenge_method=S256`; /** - * Reassemble the copy-URL rows for `label` into a single string, mirroring what - * a browser would produce when a multi-row selection is pasted into its - * address bar (whitespace stripped between chunks). Returns "" if the label - * row isn't found. + * Reassemble the copy-URL rows for `label` into a single string, mirroring a + * real multi-row terminal selection pasted into an address bar: browsers + * strip the newlines, but any other leading bytes survive (verbatim or + * percent-encoded) — so chunks are concatenated RAW, with no indent-stripping + * that could mask a corrupting prefix. Returns "" if the label row isn't + * found. */ function reassembleUrl(plainLines: string[], label: string): string { const start = plainLines.findIndex(line => line.startsWith(` ${label}`)); @@ -36,15 +38,13 @@ function reassembleUrl(plainLines: string[], label: string): string { // Inline form contains the whole URL on the label row. const inlineMatch = first.match(new RegExp(`^ ${label.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")} (.*)$`)); if (inlineMatch) return inlineMatch[1]!; - // Wrapped form: label alone, then continuation rows with a single-space indent. + // Wrapped form: indented label row, then UNINDENTED continuation chunks. + // Any indented row (the next label) or blank row ends this URL's chunks. let joined = ""; for (let i = start + 1; i < plainLines.length; i++) { const line = plainLines[i]!; - // Continuation rows have exactly one leading space; a different indent - // or label ends this URL's rows. - if (!line.startsWith(" ") || line.startsWith(" ") || line.trim().length === 0) break; - if (line.includes(":") && /^ [A-Z][^:]*:/.test(line)) break; // next label row - joined += line.slice(1); + if (line.startsWith(" ") || line.trim().length === 0) break; + joined += line; } return joined; } @@ -89,10 +89,12 @@ describe("MCPAuthorizationLinkPrompt", () => { expect(visibleWidth(line)).toBeLessThanOrEqual(width); } - // Wrapping puts the label on its own row followed by continuation - // chunks with a single-space indent. + // Wrapping puts the label on its own indented row followed by + // UNINDENTED continuation chunks — zero leading bytes, so a multi-row + // selection pastes back to the exact URL. const labelRow = plainLines.indexOf(` ${COPY_URL_LABEL}`); expect(labelRow).toBeGreaterThanOrEqual(0); + expect(plainLines[labelRow + 1]!.startsWith(" ")).toBe(false); // Chunks reassemble to the URL byte-for-byte — the trailing // `code_challenge_method=S256` MUST be present. @@ -140,7 +142,7 @@ describe("MCPAuthorizationLinkPrompt", () => { it("floors the wrap width so degenerately-narrow viewports still emit every character", () => { // Below 16 cols the terminal is unusable, but the render still emits - // chunks (bounded at 16 - indent = 15 chars). No character is silently + // chunks (bounded at the 16-column floor). No character is silently // dropped; the user can widen and reflow. const lines = new MCPAuthorizationLinkPrompt(LINEAR_AUTH_URL).render(4); const plainLines = lines.map(line => stripVTControlCharacters(line)); From 34607818ede3cf8894c3089c45e49fc44de68b7f Mon Sep 17 00:00:00 2001 From: Dylan Bohlender Date: Sun, 5 Jul 2026 16:15:09 -0600 Subject: [PATCH 33/91] docs(changelog): add Unreleased entry for the Windows opener fix --- packages/coding-agent/CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 702ba1d2e..38a06a81e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Windows browser-launch failures being unobservable: the opener now uses `%SystemRoot%`-resolved PowerShell `Start-Process` (via `-EncodedCommand`) instead of `rundll32`, which exits 0 unconditionally. Failures ShellExecute itself reports — missing target, no handler executable, access denied — now surface as non-zero exits and are logged; the encoded payload also keeps OAuth query strings (`&`-bearing) opaque to shell metacharacter parsing. + ## [16.3.8] - 2026-07-05 ### Fixed From f6e191844f80c022ca3f9ab8cb119d0240b0e3ff Mon Sep 17 00:00:00 2001 From: Dylan Bohlender Date: Sun, 5 Jul 2026 16:15:55 -0600 Subject: [PATCH 34/91] docs(changelog): add Unreleased entries for launch-URL and copy-chunk fixes --- packages/ai/CHANGELOG.md | 4 ++++ packages/coding-agent/CHANGELOG.md | 4 ++++ 2 files changed, 8 insertions(+) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d49636d82..c0cda52ce 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OAuth `launchUrl` advertisement for flows whose redirect never returns to the local callback server: custom-scheme redirects (e.g. GitLab Duo's `vscode://` URI, which `new URL` parses without complaint) and fixed non-loopback hosts no longer receive a `http://localhost:/launch` copy target that misrepresents the callback endpoint and resolves nowhere for remote users. + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 702ba1d2e..5c61c476c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed wrapped OAuth copy-URL rows corrupting on paste: continuation chunks no longer carry a leading indent, so a multi-row terminal selection reassembles to the exact authorize URL (browsers strip newlines on paste but preserve or percent-encode embedded spaces, which previously corrupted the URL at every chunk boundary). + ## [16.3.8] - 2026-07-05 ### Fixed From 78d4978c517ee85744605e1d5a9f172a4cfc3949 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Thu, 25 Jun 2026 15:43:27 -0700 Subject: [PATCH 35/91] fix(bash): support disabled command deadlines Treat timeout 0 as an explicit no-deadline contract across the bash tool, executor, async job, and PTY paths. Signed-off-by: Christian Stewart --- .../coding-agent/src/exec/bash-executor.ts | 23 +++--- .../coding-agent/src/prompts/tools/bash.md | 6 +- .../src/tools/bash-interactive.ts | 2 +- packages/coding-agent/src/tools/bash.ts | 79 ++++++++++++------- .../coding-agent/test/bash-executor.test.ts | 9 +++ packages/coding-agent/test/tools.test.ts | 15 ++++ 6 files changed, 92 insertions(+), 42 deletions(-) diff --git a/packages/coding-agent/src/exec/bash-executor.ts b/packages/coding-agent/src/exec/bash-executor.ts index dfdcee8a3..fc7d7dc69 100644 --- a/packages/coding-agent/src/exec/bash-executor.ts +++ b/packages/coding-agent/src/exec/bash-executor.ts @@ -14,6 +14,7 @@ import { buildNonInteractiveEnv } from "./non-interactive-env"; export interface BashExecutorOptions { cwd?: string; + /** Milliseconds before aborting the command; 0 disables the executor deadline. */ timeout?: number; onChunk?: (chunk: string) => void; chunkThrottleMs?: number; @@ -296,11 +297,15 @@ export async function executeBash(command: string, options?: BashExecutorOptions let timeoutTimer: NodeJS.Timeout | undefined; const timeoutDeferred = Promise.withResolvers<"timeout">(); - const baseTimeoutMs = Math.max(1_000, options?.timeout ?? 300_000); - timeoutTimer = setTimeout(() => { - abortCurrentExecution(); - timeoutDeferred.resolve("timeout"); - }, baseTimeoutMs); + const requestedTimeoutMs = options?.timeout; + const deadlineTimeoutMs = requestedTimeoutMs === 0 ? undefined : Math.max(1_000, requestedTimeoutMs ?? 300_000); + const nativeTimeoutMs = requestedTimeoutMs !== undefined && requestedTimeoutMs > 0 ? requestedTimeoutMs : undefined; + if (deadlineTimeoutMs !== undefined) { + timeoutTimer = setTimeout(() => { + abortCurrentExecution(); + timeoutDeferred.resolve("timeout"); + }, deadlineTimeoutMs); + } let resetSession = false; @@ -311,7 +316,7 @@ export async function executeBash(command: string, options?: BashExecutorOptions command: finalCommand, cwd: commandCwd, env: commandEnv, - timeoutMs: options?.timeout, + timeoutMs: nativeTimeoutMs, signal: runAbortController.signal, }, (err, chunk) => { @@ -328,7 +333,7 @@ export async function executeBash(command: string, options?: BashExecutorOptions sessionEnv: shellEnv, snapshotPath: snapshotPath ?? undefined, minimizer, - timeoutMs: options?.timeout, + timeoutMs: nativeTimeoutMs, signal: runAbortController.signal, }, (err, chunk) => { @@ -359,8 +364,8 @@ export async function executeBash(command: string, options?: BashExecutorOptions exitCode: undefined, cancelled: true, ...(await sink.dump( - winner.kind === "timeout" - ? `Command timed out after ${Math.round(baseTimeoutMs / 1000)} seconds` + winner.kind === "timeout" && deadlineTimeoutMs !== undefined + ? `Command timed out after ${Math.round(deadlineTimeoutMs / 1000)} seconds` : "Command cancelled", )), }; diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 04bcca23c..ddc0f7b90 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -41,9 +41,9 @@ Anything below → `eval` cell, not bash: {{#if asyncEnabled}} # Timeout and async -- `timeout` is seconds, clamped to `1..3600`; the process is killed on elapse. -- `async: true` defers only reporting — it does NOT extend the timeout; a daemon with `async: true` is still killed at the clamped timeout. -- Need >3600s? Detach/manage lifecycle yourself (`cmd &`, supervisor, self-restarting script). The shell session persists across calls. +- `timeout` is seconds; nonzero values are clamped to `1..3600` and the process is killed on elapse. Set `timeout: 0` only for commands that must run until completion or explicit cancellation. +- `async: true` defers only reporting — it does NOT extend a nonzero timeout; use `timeout: 0` when a daemon or watcher must be cancellation-owned. +- Need a daemon or >3600s run? Use `async: true` with `timeout: 0` when the harness should keep it alive until cancellation, or detach/manage lifecycle yourself (`cmd &`, supervisor, self-restarting script). The shell session persists across calls. {{/if}} {{#if autoBackgroundEnabled}} diff --git a/packages/coding-agent/src/tools/bash-interactive.ts b/packages/coding-agent/src/tools/bash-interactive.ts index 3c1a57e6c..fc39cb0c8 100644 --- a/packages/coding-agent/src/tools/bash-interactive.ts +++ b/packages/coding-agent/src/tools/bash-interactive.ts @@ -300,7 +300,7 @@ export async function runInteractiveBashPty( options: { command: string; cwd: string; - timeoutMs: number; + timeoutMs?: number; signal?: AbortSignal; env?: Record; artifactPath?: string; diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index ad8d8faff..0a57890b0 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -131,7 +131,7 @@ async function saveBashOriginalArtifact(session: ToolSession, originalText: stri } } -const BASH_TIMEOUT_DESCRIPTION = `timeout in seconds; clamped to ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}`; +const BASH_TIMEOUT_DESCRIPTION = `timeout in seconds; 0 disables the command deadline; nonzero values are clamped to ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}`; const bashSchemaBase = type({ command: type("string").describe("command to execute"), @@ -166,6 +166,7 @@ export interface BashToolDetails { meta?: OutputMeta; timeoutSeconds?: number; requestedTimeoutSeconds?: number; + timeoutDisabled?: boolean; wallTimeMs?: number; /** Exit code of a command that ran to completion but failed (non-zero). */ exitCode?: number; @@ -421,7 +422,11 @@ export class BashTool implements AgentTool { const details: BashToolDetails = { - timeoutSeconds: timeoutSec, async: { state: "running", jobId, type: "bash" }, }; + if (timeoutSec === undefined) { + details.timeoutDisabled = true; + } else { + details.timeoutSeconds = timeoutSec; + } if (options.requestedTimeoutSec !== undefined && options.requestedTimeoutSec !== timeoutSec) { details.requestedTimeoutSeconds = options.requestedTimeoutSec; } @@ -539,8 +551,8 @@ export class BashTool implements AgentTool ({ kind: "timeout" as const })); + const timeoutPromise = timeoutMs + ? Bun.sleep(timeoutMs).then(() => ({ kind: "timeout" as const })) + : undefined; // Poll until the process exits, times out, or the caller aborts. for (;;) { const racers: Array> = [ exitPromise.then(s => ({ kind: "exit" as const, status: s })), - timeoutPromise, Bun.sleep(250).then(() => ({ kind: "poll" as const })), ]; + if (timeoutPromise) racers.push(timeoutPromise); if (signal) { racers.push(abortedP.then(() => ({ kind: "aborted" as const }))); } @@ -1053,7 +1072,7 @@ export class BashTool implements AgentTool(config: ShellRendererConfig) { const showingFullOutput = expanded && renderContext?.isFullOutput === true; // Build truncation warning - const timeoutSeconds = details?.timeoutSeconds ?? renderContext?.timeout; + const timeoutDisabled = details?.timeoutDisabled === true || renderContext?.timeout === 0; + const timeoutSeconds = timeoutDisabled ? undefined : (details?.timeoutSeconds ?? renderContext?.timeout); const requestedTimeoutSeconds = details?.requestedTimeoutSeconds; const wallTimeMs = details?.wallTimeMs; const statsParts: string[] = []; if (wallTimeMs !== undefined) { statsParts.push(`Wall: ${formatWallTimeSeconds(wallTimeMs)}s`); } + if (timeoutDisabled) { + statsParts.push("Timeout: disabled"); + } if (typeof timeoutSeconds === "number") { statsParts.push( requestedTimeoutSeconds !== undefined && requestedTimeoutSeconds !== timeoutSeconds diff --git a/packages/coding-agent/test/bash-executor.test.ts b/packages/coding-agent/test/bash-executor.test.ts index c8436d7da..2e32fd4ae 100644 --- a/packages/coding-agent/test/bash-executor.test.ts +++ b/packages/coding-agent/test/bash-executor.test.ts @@ -402,6 +402,15 @@ exit 64 expect(result.output).not.toContain("done"); }); + it("does not arm a deadline when timeout is zero", async () => { + if (process.platform === "win32") { + return; + } + const result = await executeBash("sleep 0.1; echo done", { cwd: tempDir, timeout: 0 }); + expect(result.cancelled).toBe(false); + expect(result.output.trim()).toBe("done"); + }); + it("aborts commands", async () => { if (process.platform === "win32") { return; diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index a556a4351..a3baf9bbd 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -1492,6 +1492,21 @@ function b() { expect(result.details?.requestedTimeoutSeconds).toBe(7200); }); + it("should disable the command deadline when timeout is zero", async () => { + vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.05); + + const result = await bashTool.execute("test-call-timeout-disabled", { + command: "printf 'start\\n'; sleep 0.1; printf 'done\\n'", + timeout: 0, + }); + + const output = getTextOutput(result); + expect(output).toContain("start"); + expect(output).toContain("done"); + expect(result.details?.timeoutDisabled).toBe(true); + expect(result.details?.timeoutSeconds).toBeUndefined(); + }); + it("should respect timeout", async () => { // Reduce the effective timeout through the production clamp seam; the // real subprocess kill-on-timeout path is still exercised, just faster. From 722a06abce669a3fcc18e50f1e93376dfaa5fdc4 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Tue, 30 Jun 2026 20:19:50 -0700 Subject: [PATCH 36/91] fix(coding-agent): use local date in system prompt Signed-off-by: Christian Stewart --- .../coding-agent/src/session/agent-session.ts | 3 +- packages/coding-agent/src/system-prompt.ts | 3 +- packages/coding-agent/src/utils/local-date.ts | 7 +++++ .../agent-session-tool-rebuild-skip.test.ts | 30 ++++++++++++------- .../test/system-prompt-model.test.ts | 29 +++++++++++++++++- 5 files changed, 58 insertions(+), 14 deletions(-) create mode 100644 packages/coding-agent/src/utils/local-date.ts diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index da4492fc4..5e49cba05 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -314,6 +314,7 @@ import { resolveFileDisplayMode } from "../utils/file-display-mode"; import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions"; import { normalizeModelContextImages } from "../utils/image-loading"; import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback"; +import { formatLocalCalendarDate } from "../utils/local-date"; import { generateSessionTitle } from "../utils/title-generator"; import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice"; import type { AuthStorage } from "./auth-storage"; @@ -6517,7 +6518,7 @@ export class AgentSession { entries.sort(); instructionsSegment = entries.join("\u0006"); } - const date = new Date().toISOString().slice(0, 10); + const date = formatLocalCalendarDate(); return `${nameSegment}\u0003${descriptionSegment}\u0005${registrySegment}\u0007${instructionsSegment}|${date}`; } diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 890e37484..196be0dee 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -24,6 +24,7 @@ import projectPromptTemplate from "./prompts/system/project-prompt.md" with { ty import systemPromptTemplate from "./prompts/system/system-prompt.md" with { type: "text" }; import { shortenPath } from "./tools/render-utils"; import { type ActiveRepoContext, resolveActiveRepoContext } from "./utils/active-repo-context"; +import { formatLocalCalendarDate } from "./utils/local-date"; import { normalizePromptPath } from "./utils/prompt-path"; import { AGENTS_MD_LIMIT, buildWorkspaceTree, type WorkspaceTree } from "./workspace-tree"; @@ -677,7 +678,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): } } - const date = new Date().toISOString().slice(0, 10); + const date = formatLocalCalendarDate(); const dateTime = date; const promptCwd = shortenPath(normalizePromptPath(resolvedCwd)); const activeRepoContextPrompt = renderActiveRepoContextPrompt(activeRepoContext); diff --git a/packages/coding-agent/src/utils/local-date.ts b/packages/coding-agent/src/utils/local-date.ts new file mode 100644 index 000000000..80962f102 --- /dev/null +++ b/packages/coding-agent/src/utils/local-date.ts @@ -0,0 +1,7 @@ +/** formatLocalCalendarDate formats a Date as YYYY-MM-DD in the host local timezone. */ +export function formatLocalCalendarDate(date: Date = new Date()): string { + const year = date.getFullYear(); + const month = String(date.getMonth() + 1).padStart(2, "0"); + const day = String(date.getDate()).padStart(2, "0"); + return `${year}-${month}-${day}`; +} diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 6cefd9c5b..53f795f2c 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -388,11 +388,14 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { await session.refreshMCPTools([dynamicTool]); expect(rebuildCount).toBe(baseline + 1); }); - it("rebuilds when the calendar date rolls over between tool-stable MCP refreshes", async () => { - // `buildSystemPrompt` injects today's date into the prompt body. - // A session spanning midnight must not serve yesterday's date after an MCP - // reconnect that happens to bring an identical tool set. - setSystemTime(new Date("2025-01-01T23:59:58Z")); + it("rebuilds when the local calendar date rolls over between tool-stable MCP refreshes", async () => { + // `buildSystemPrompt` injects today's date into the prompt body. A session + // spanning local midnight must not serve yesterday's date after an MCP + // reconnect that happens to bring an identical tool set, even when UTC has + // not rolled over. + const originalTimezone = process.env.TZ; + process.env.TZ = "America/Los_Angeles"; + setSystemTime(new Date("2026-07-01T06:59:58Z")); try { let rebuildCount = 0; const { session } = newSession(async toolNames => { @@ -405,22 +408,27 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { await session.refreshMCPTools([tool]); expect(rebuildCount).toBe(1); - // Same tools, same day: signature matches, skip. + // Same tools, same local day: signature matches, skip. await session.refreshMCPTools([tool]); expect(rebuildCount).toBe(1); - // Advance past midnight. - setSystemTime(new Date("2025-01-02T00:00:01Z")); + // Advance past local midnight while the UTC date remains 2026-07-01. + setSystemTime(new Date("2026-07-01T07:00:01Z")); - // Same tools, new calendar day: date segment changed, must rebuild. + // Same tools, new local calendar day: date segment changed, must rebuild. await session.refreshMCPTools([tool]); expect(rebuildCount).toBe(2); - // Same tools, same new day: skip again. + // Same tools, same new local day: skip again. await session.refreshMCPTools([tool]); expect(rebuildCount).toBe(2); } finally { - setSystemTime(); // restore real time + setSystemTime(); + if (originalTimezone === undefined) { + delete process.env.TZ; + } else { + process.env.TZ = originalTimezone; + } } }); it("does not rebuild when MCP server instructions change only beyond the 4000-char truncation boundary", async () => { diff --git a/packages/coding-agent/test/system-prompt-model.test.ts b/packages/coding-agent/test/system-prompt-model.test.ts index 8a0eba4f3..3f253052e 100644 --- a/packages/coding-agent/test/system-prompt-model.test.ts +++ b/packages/coding-agent/test/system-prompt-model.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, setSystemTime } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -49,6 +49,33 @@ describe("system prompt model identifier", () => { expect(systemPrompt.join("\n\n")).toContain("Model: anthropic/claude-opus-4"); }); + it("renders the prompt date from the local timezone rather than UTC", async () => { + const originalTimezone = process.env.TZ; + process.env.TZ = "America/Los_Angeles"; + setSystemTime(new Date("2026-07-01T03:15:00Z")); + try { + const { systemPrompt } = await buildSystemPrompt({ + cwd: tempDir, + contextFiles: [], + skills: [], + rules: [], + toolNames: [], + workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, + }); + const rendered = systemPrompt.join("\n\n"); + + expect(rendered).toContain("Today is 2026-06-30"); + expect(rendered).not.toContain("Today is 2026-07-01"); + } finally { + setSystemTime(); + if (originalTimezone === undefined) { + delete process.env.TZ; + } else { + process.env.TZ = originalTimezone; + } + } + }); + it("omits the model line when no model is provided", async () => { const { systemPrompt } = await buildSystemPrompt({ cwd: tempDir, From b765657f28e292f3f66304d2293611563a67a69c Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 28 Jun 2026 01:30:51 -0700 Subject: [PATCH 37/91] fix(prompting): hide eval guidance when disabled Stop advertising eval in the default prompt and workflow notice when no eval backend is enabled. Gate bash guidance on live eval backend availability and cover the disabled-backend rendering contract. Agent-Milestone: tooling: hide eval prompt guidance when eval backends are disabled Signed-off-by: Christian Stewart --- .../src/prompts/system/system-prompt.md | 3 +- .../src/prompts/system/workflow-notice.md | 10 ++-- .../coding-agent/src/prompts/tools/bash.md | 17 +++++-- packages/coding-agent/src/system-prompt.ts | 2 +- packages/coding-agent/src/tools/bash.ts | 3 ++ .../test/system-prompt-inventory.test.ts | 50 ++++++++++++++++++- 6 files changed, 73 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index e852b653a..e90d57c5e 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -110,9 +110,8 @@ You MUST use the specialized tool over its shell equivalent: {{#has tools "lsp"}}- Code intelligence → `{{toolRefs.lsp}}`.{{/has}} {{#has tools "grep"}}- Regex search → `{{toolRefs.grep}}`, not `grep`, `rg`, or `awk`.{{/has}} {{#has tools "glob"}}- Globbing → `{{toolRefs.glob}}`, not `ls **/*.ext` or `fd`.{{/has}} -{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(…)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. NEVER write multiline or inline-script bash.{{/has}} {{#has tools "bash"}}- `{{toolRefs.bash}}`: real binaries and short fact pipelines only. Commands shadowing the specialized tools above are blocked.{{/has}} -{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash.{{#has tools "eval"}} Needs control flow, state, or fights shell quoting → `{{toolRefs.eval}}`.{{/has}} Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} +{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash. Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} {{#has tools "report_tool_issue"}} diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 4071c9df5..0ed092a44 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -1,8 +1,8 @@ -The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Author the orchestration as Python in the `eval` tool and fan out subagents — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. +The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Use the `task` tool for batched fan-out — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. -Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline FIRST (list the files, scope the diff, find the call sites) to discover the work-list, then fan out over it — you don't need to know the shape before the *task*, only before the *fan-out*. Common shapes, each a well-scoped `eval` call you can chain across turns: +Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline FIRST (list the files, scope the diff, find the call sites) to discover the work-list, then fan out over it — you don't need to know the shape before the *task*, only before the *fan-out*. Common shapes: - **Understand** — parallel readers over subsystems → structured map - **Design** — judge panel of N independent approaches → scored synthesis - **Review** — split into dimensions → find per dimension → adversarially verify each finding @@ -60,11 +60,11 @@ Compose the harness the task calls for: Scale to the ask: "find any bugs" → a few finders, single-vote verify. "thoroughly audit / be comprehensive" → larger finder pool, 3–5-vote adversarial pass, a synthesis stage. - - Decompose the surface first; capture it in `todo` when it spans phases. -- Prefer `schema=` for any agent whose output you branch on. -- After a fan-out returns, YOU own correctness: read the artifacts, run the gate, verify before acting. Subagents do the legwork; they don't get the last word. +- Batch independent subagents in one `task` call when the available `task` schema supports batching; otherwise issue independent task calls in the same assistant turn. +- Give every subagent a narrow target, explicit non-goals, and a concrete return packet. Shared background goes in a `local://` file referenced from each prompt, not pasted repeatedly. +- After fan-out returns, YOU own correctness: read the artifacts, run the gate, verify before acting. Subagents do the legwork; they don't get the last word. - Keep going until the task is closed — a returned fan-out is a step, not a stopping point. diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 04bcca23c..5d4fb8d69 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -6,14 +6,22 @@ The shell invokes **real binaries** with simple args. It is NOT full GNU Bash. Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact and does not depend on shell-specific regex/quoting (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`). -Anything below → `eval` cell, not bash: +{{#if hasEval}}Anything below → `eval` cell, not bash: - Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language - Heredocs (`< - `cwd` sets the working dir, not `cd dir && …` @@ -30,7 +38,10 @@ Anything below → `eval` cell, not bash: -- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps. +{{#if hasEval}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.{{else}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a shell program; use a purpose-built tool or checked-in script instead.{{/if}} +- NEVER shell out to search content or files: `grep/rg` → `grep`. +- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing. +- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 890e37484..a22d957a0 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -388,7 +388,7 @@ export async function loadSystemPromptFiles(options: LoadContextFilesOptions = { return userLevel?.content ?? null; } -export const DEFAULT_SYSTEM_PROMPT_TOOL_NAMES = ["read", "bash", "eval", "edit", "write"] as const; +export const DEFAULT_SYSTEM_PROMPT_TOOL_NAMES = ["read", "bash", "edit", "write"] as const; export interface SystemPromptToolMetadata { label: string; diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index ad8d8faff..aace6a3bb 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -27,6 +27,7 @@ import { type BashInteractiveResult, runInteractiveBashPty } from "./bash-intera import { checkBashInterception } from "./bash-interceptor"; import { canUseInteractiveBashPty } from "./bash-pty-selection"; import { expandInternalUrls, type InternalUrlExpansionOptions } from "./bash-skill-urls"; +import { resolveEvalBackends } from "./eval-backends"; import { invalidateGithubCacheForBashCommand } from "./gh-cache-invalidation"; import { formatStyledTruncationWarning, @@ -397,6 +398,7 @@ export class BashTool implements AgentTool { return text.slice(inventoryStart, inventoryEnd); } + function makeToolSession(settings: Settings): ToolSession { + return { + cwd: tempDir, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings, + } as ToolSession; + } + it("renders a compact name list only when native tools are active and descriptors stay in schemas", async () => { const text = await render({ nativeTools: true, inlineToolDescriptors: false }); expect(text).toContain("- Read: `read`"); @@ -132,6 +144,42 @@ describe("system prompt tool inventory", () => { } expect(inventory).not.toContain("- `browser`"); expect(inventory).not.toContain("- `task`"); + expect(inventory).not.toContain("- `eval`"); + }); + + it("omits eval prompt guidance when every eval backend is disabled", async () => { + const settings = Settings.isolated({ + "eval.py": false, + "eval.js": false, + "eval.rb": false, + "eval.jl": false, + }); + const session = makeToolSession(settings); + const tools = await createTools(session, ["bash", "eval"]); + const toolNames = tools.map(tool => tool.name); + const bash = tools.find(tool => tool.name === "bash"); + + expect(toolNames).toContain("bash"); + expect(toolNames).not.toContain("eval"); + expect(bash?.description).toContain("purpose-built tool"); + expect(bash?.description).not.toContain("eval` cell"); + expect(bash?.description).not.toContain("use `eval` cells"); + + const { systemPrompt } = await buildSystemPrompt({ + cwd: tempDir, + contextFiles: [], + skills: [], + rules: [], + toolNames, + tools: buildSystemPromptToolMetadata(new Map(tools.map(tool => [tool.name, tool]))), + workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, + nativeTools: true, + inlineToolDescriptors: true, + }); + const text = systemPrompt.join("\n\n"); + + expect(text).not.toContain("Default for any compute"); + expect(text).not.toContain("use `eval` cells"); }); it("SDK wrapper renders provided tools instead of the fallback inventory", async () => { From 22279bc03b046c856a4d312b3452dc12f61a8fe3 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Thu, 25 Jun 2026 13:00:32 -0700 Subject: [PATCH 38/91] fix: cramped slash autocomplete menu --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../src/config/settings-schema.ts | 3 +-- packages/tui/CHANGELOG.md | 4 ++++ packages/tui/src/components/editor.ts | 3 +-- packages/tui/test/editor.test.ts | 24 +++++++++++++++++++ 5 files changed, 34 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 702ba1d2e..a939e8b83 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- Increased the default autocomplete dropdown from 5 to 10 visible items so slash command menus are less cramped. + ## [16.3.8] - 2026-07-05 ### Fixed diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 00af4ed69..da512d952 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -1493,7 +1493,7 @@ export const SETTINGS_SCHEMA = { autocompleteMaxVisible: { type: "number", - default: 5, + default: 10, ui: { tab: "interaction", group: "Input", @@ -1502,7 +1502,6 @@ export const SETTINGS_SCHEMA = { options: [ { value: "3", label: "3 items" }, { value: "5", label: "5 items" }, - { value: "7", label: "7 items" }, { value: "10", label: "10 items" }, { value: "15", label: "15 items" }, { value: "20", label: "20 items" }, diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index b99d781d7..8c01a59cd 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Kept slash command autocomplete rows compact by truncating descriptions instead of wrapping them into multi-line blocks, and raised the editor's default autocomplete window to 10 items. + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 2eb353fd3..4d88f266f 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -31,7 +31,6 @@ const SLASH_COMMAND_SELECT_LIST_LAYOUT: SelectListLayoutOptions = { minPrimaryColumnWidth: 12, maxPrimaryColumnWidth: 32, overflowSearch: false, - wrapDescription: true, }; function sanitizeLoadedText(text: string): string { @@ -423,7 +422,7 @@ export class Editor implements Component, Focusable { #autocompleteState: "regular" | "force" | null = null; #autocompletePrefix: string = ""; #autocompleteRequestId: number = 0; - #autocompleteMaxVisible: number = 5; + #autocompleteMaxVisible: number = 10; onAutocompleteUpdate?: () => void; // Paste tracking for large pastes diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index d451c758e..9240aebb0 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -309,6 +309,30 @@ describe("Editor component", () => { await expect(promise).resolves.toBe("/"); }); + it("renders slash-command suggestions as compact item rows", async () => { + const editor = new Editor(defaultEditorTheme); + const longDescription = + "Plan and execute non-trivial architectural improvements to the codebase without turning each slash command into a multi-line block."; + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider( + Array.from({ length: 12 }, (_, i) => ({ + name: `cmd${i}`, + description: longDescription, + })), + "/tmp", + ), + ); + + editor.handleInput("/"); + await Bun.sleep(0); + + const rendered = editor.render(80).map(line => stripVTControlCharacters(line)); + for (let i = 0; i < 10; i += 1) { + expect(rendered.some(line => line.includes(`cmd${i}`))).toBe(true); + } + expect(rendered.some(line => line.includes("cmd10"))).toBe(false); + }); + it("triggers file-reference autocomplete when typing at-sign", async () => { const editor = new Editor(defaultEditorTheme); const { promise, resolve } = Promise.withResolvers(); From 92765fc409a956c520617d6291d2c367d5ded3ac Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 16:59:54 -0700 Subject: [PATCH 39/91] fix(prompting): align workflow and bash guidance with active tools Signed-off-by: Christian Stewart --- .../src/prompts/system/workflow-notice.md | 99 +++++++++---------- .../coding-agent/src/prompts/tools/bash.md | 6 +- packages/coding-agent/src/tools/bash.ts | 12 ++- packages/coding-agent/src/tools/index.ts | 10 +- .../coding-agent/test/modes/workflow.test.ts | 5 +- .../test/system-prompt-inventory.test.ts | 3 + 6 files changed, 73 insertions(+), 62 deletions(-) diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 0ed092a44..3a12bb8ce 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -2,69 +2,66 @@ The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Use the `task` tool for batched fan-out — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. -Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline FIRST (list the files, scope the diff, find the call sites) to discover the work-list, then fan out over it — you don't need to know the shape before the *task*, only before the *fan-out*. Common shapes: -- **Understand** — parallel readers over subsystems → structured map -- **Design** — judge panel of N independent approaches → scored synthesis -- **Review** — split into dimensions → find per dimension → adversarially verify each finding -- **Research** — multi-modal sweep → deep-read the hits → synthesize -- **Migrate** — discover sites → transform each → verify +Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline first (list the files, scope the diff, find the call sites) to discover the work list, then fan out over it. Common shapes: +- **Understand** — parallel readers over subsystems → structured map. +- **Design** — independent approaches → scored synthesis. +- **Review** — split dimensions → find per dimension → adversarially verify each finding. +- **Research** — multi-modal sweep → deep-read the hits → synthesize. +- **Migrate** — discover sites → transform each → verify. - -State persists across eval calls, so scout in one call and fan out in the next. Every eval call has: + +Call `task` once per independent fan-out batch. Put shared background in `context`, and put each independent work item in `tasks[]`. Do not emulate batching with shell loops or eval helper APIs. -- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes. Recursion follows `task.maxRecursionDepth` (default 2; `-1` uses eval's hard cap 3): main agent depth = 0, each `agent()` child increments depth by 1, and a spawner may call `agent()` only while its current `taskDepth < effective cap`. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts. -- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. -- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`. -- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. -- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it. -- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget. +`context` must carry the shared contract: -Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase. - + # Goal + What the batch accomplishes. + # Constraints + Rules, non-goals, permissions, and verification limits. + # Contract + Shared interfaces, output shape, branch/base assumptions, and coordination rules. + +Each task assignment must be self-contained: + + # Target + Exact files, symbols, subsystem, or evidence surface; explicit non-goals. + # Change + What to inspect or modify, step by step, including APIs and patterns to reuse. + # Acceptance + Observable result, return packet, and local verification. Subagents skip formatters, + linters, and project-wide tests; the parent runs shared proof once. + +Use specific roles (`Storage Reviewer`, `CLI Migrator`, `Security Skeptic`) rather than generic workers. Dispatch at most one layer unless the user grants a named second-layer purpose. + -For independent per-item chains (review → verify, fetch → extract → score), wrap the WHOLE chain in one function and run it with `parallel()` — then each item flows through its own steps without waiting on the others: +Decompose first, then batch the independent leaves: - DIMENSIONS = [{"key": "bugs", "prompt": "…"}, {"key": "perf", "prompt": "…"}] - def review_and_verify(d): - found = agent(d["prompt"], label=f"review:{d['key']}", schema=FINDINGS_SCHEMA) - return parallel([lambda f=f: {**f, "verdict": agent( - f"Refute if you can (default refuted when unsure): {f['title']}", - label=f"verify:{f['file']}", schema=VERDICT_SCHEMA)} for f in found["findings"]]) - phase("Review") - results = parallel([lambda d=d: review_and_verify(d) for d in DIMENSIONS]) - confirmed = [f for group in results for f in group if f["verdict"]["is_real"]] + task( + context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...", + tasks: [ + { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection...\n# Acceptance\nReturn confirmed findings only..." }, + { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." }, + ] + ) -Reach for `pipeline()` only when a stage genuinely needs ALL of the previous stage first — dedup/merge across the whole set, early-exit on zero, or "compare against the other findings" — because its inter-stage barrier makes every item wait for the slowest peer: - - phase("Find") - found = parallel([lambda d=d: agent(d["prompt"], schema=FINDINGS_SCHEMA) for d in DIMENSIONS]) - findings = dedupe([f for r in found for f in r["findings"]]) # needs everything at once - phase("Verify") - verdicts = parallel([lambda f=f: agent(verify_prompt(f), schema=VERDICT_SCHEMA) for f in findings]) - -Don't add a barrier just to flatten/map/filter — do that with plain Python between calls. Nested `parallel()` pools each cap independently, so keep total fan-out sane. +Prefer one wide batch over serial subagent calls when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing. -Compose the harness the task calls for: -- **Adversarial verify** — N independent skeptics per finding, each prompted to REFUTE; keep it only if a majority survive. `votes = parallel([lambda i=i: agent(f"Refute: {claim}. refuted=true if unsure.", schema=VERDICT) for i in range(3)])`, then keep when `sum(not v["refuted"] for v in votes) ≥ 2`. -- **Perspective-diverse verify** — give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters. -- **Judge panel** — N attempts from different angles, scored by parallel judges; synthesize from the winner, graft the best of the rest. -- **Loop-until-dry** — for unknown-size discovery, keep spawning finders until K consecutive rounds surface nothing new; dedup against everything SEEN, not just what was confirmed, or it never converges. -- **Multi-modal sweep** — parallel finders each searching a different way (by-container, by-content, by-entity, by-time), each blind to the others. -- **Completeness critic** — a final agent that asks "what's missing — modality not run, claim unverified, file unread?"; its answer is the next round. -- **Budget/count loops** — `while len(bugs) < 10:` to hit a target, or `while budget.total and budget.remaining() > 50_000:` to scale depth to the turn budget; `log()` each round. -- **No silent caps** — if you bound coverage (top-N, no-retry, sampling), `log()` what you dropped; silent truncation reads as "covered everything" when it didn't. - -Scale to the ask: "find any bugs" → a few finders, single-vote verify. "thoroughly audit / be comprehensive" → larger finder pool, 3–5-vote adversarial pass, a synthesis stage. +- **Adversarial verify** — dispatch skeptical reviewers with distinct targets, then keep only findings the parent can verify against source. +- **Perspective-diverse review** — use separate correctness, security, performance, and maintainability roles instead of identical reviewers. +- **Completeness critic** — after the first batch, dispatch one read-only critic that asks what modality, file, claim, or proof was missed. +- **No silent caps** — if you bound coverage (top-N, no retry, sampling), state what was dropped and why before acting. +- **Parent owns closure** — subagents return evidence; the parent reads it, resolves contradictions, runs proof, and makes the final decision. + -- Decompose the surface first; capture it in `todo` when it spans phases. -- Batch independent subagents in one `task` call when the available `task` schema supports batching; otherwise issue independent task calls in the same assistant turn. -- Give every subagent a narrow target, explicit non-goals, and a concrete return packet. Shared background goes in a `local://` file referenced from each prompt, not pasted repeatedly. -- After fan-out returns, YOU own correctness: read the artifacts, run the gate, verify before acting. Subagents do the legwork; they don't get the last word. -- Keep going until the task is closed — a returned fan-out is a step, not a stopping point. +- Capture multi-phase workflow state in the visible todo system when available. +- Batch independent subagents in one `task` call. +- Give every subagent a narrow target, explicit non-goals, and a concrete return packet. +- After fan-out returns, read the artifacts, patch or decide, and run the shared gate. +- Keep going until the task is closed — returned fan-out is a step, not a stopping point. diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 5d4fb8d69..9af269ac3 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -21,7 +21,7 @@ Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a f - Multiline commands, `&&`-chains mixing control flow - Quote/JSON escaping that fights the shell {{/if}} -- GNU grep BRE extensions are not guaranteed in the embedded shell: use `grep -E 'json|tool'` for alternation instead of `grep 'json\|tool'`; use the built-in `grep` tool with `pattern: "json|tool"` (Rust regex, so `\bword\b` works there){{#if hasEval}}, or `eval` for exact text processing{{/if}}. +{{#if hasGrep}}- GNU grep BRE extensions are not guaranteed in the embedded shell: use `grep -E 'json|tool'` for alternation instead of `grep 'json\|tool'`; use the built-in `grep` tool with `pattern: "json|tool"` (Rust regex, so `\bword\b` works there){{#if hasEval}}, or `eval` for exact text processing{{/if}}.{{else}}- GNU grep BRE extensions are not guaranteed in the embedded shell: use `grep -E 'json|tool'` for alternation instead of `grep 'json\|tool'`{{#if hasEval}}, or use `eval` for exact text processing{{/if}}.{{/if}} - `cwd` sets the working dir, not `cd dir && …` @@ -39,8 +39,8 @@ Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a f {{#if hasEval}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.{{else}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a shell program; use a purpose-built tool or checked-in script instead.{{/if}} -- NEVER shell out to search content or files: `grep/rg` → `grep`. -- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing. +{{#if hasGrep}}- NEVER shell out to search content or files: `grep/rg` → `grep`.{{else}}- Avoid shelling out for broad content search; use an active search/read tool when one is available.{{/if}} +{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing.{{else}}- Prefer `read` for known file and directory reads. Only use shell listing when no file-listing tool is active.{{/if}}{{else}}{{#if hasGlob}}- Prefer `glob` for file discovery; avoid `find` when `glob` is active.{{else}}- If no file read/listing tool is active, keep shell inspection narrow and state that limitation.{{/if}}{{/if}} - Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index aace6a3bb..9feef6a76 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -399,15 +399,17 @@ export class BashTool implements AgentTool this.session.isToolActive?.(name) ?? fallback; this.description = prompt.render(bashDescription, { asyncEnabled: this.#asyncEnabled, autoBackgroundEnabled: this.#autoBackgroundEnabled, autoBackgroundThresholdSeconds: Math.max(0, Math.floor(this.#autoBackgroundThresholdMs / 1000)), - hasAstGrep: this.session.settings.get("astGrep.enabled"), - hasAstEdit: this.session.settings.get("astEdit.enabled"), - hasGrep: this.session.settings.get("grep.enabled"), - hasGlob: this.session.settings.get("glob.enabled"), - hasEval: evalBackends.python || evalBackends.js || evalBackends.ruby || evalBackends.julia, + hasAstGrep: isToolActive("ast_grep", this.session.settings.get("astGrep.enabled")), + hasAstEdit: isToolActive("ast_edit", this.session.settings.get("astEdit.enabled")), + hasGrep: isToolActive("grep", this.session.settings.get("grep.enabled")), + hasGlob: isToolActive("glob", this.session.settings.get("glob.enabled")), + hasRead: isToolActive("read", true), + hasEval: isToolActive("eval", evalBackends.python || evalBackends.js || evalBackends.ruby || evalBackends.julia), }); } diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index f42e49741..621c80f86 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -224,6 +224,8 @@ export interface ToolSession { getAgentId?: () => string | null; /** Look up a registered tool by name (used by the eval js backend's tool bridge). */ getToolByName?: (name: string) => AgentTool | undefined; + /** Return whether a built-in tool is active in this turn's tool set. */ + isToolActive?: (name: string) => boolean; /** Agent registry for IRC routing across live sessions. */ agentRegistry?: AgentRegistry; /** Get artifacts directory for artifact:// URLs */ @@ -647,9 +649,15 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P ...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []), ]; + const activeToolNames = new Set(baseEntries.map(([name]) => name)); + const toolFactorySession: ToolSession = { + ...session, + isToolActive: name => activeToolNames.has(name), + }; + const baseResults = await Promise.all( baseEntries.map(async ([name, factory]) => { - const tool = await logger.time(`createTools:${name}`, factory as ToolFactory, session); + const tool = await logger.time(`createTools:${name}`, factory as ToolFactory, toolFactorySession); return tool ? wrapToolWithMetaNotice(tool) : null; }), ); diff --git a/packages/coding-agent/test/modes/workflow.test.ts b/packages/coding-agent/test/modes/workflow.test.ts index 3d23c40b3..a9e719b5b 100644 --- a/packages/coding-agent/test/modes/workflow.test.ts +++ b/packages/coding-agent/test/modes/workflow.test.ts @@ -48,9 +48,10 @@ describe("workflow keyword highlighting", () => { }); describe("workflow notice", () => { - it("is a non-empty system notice carrying the eval-fan-out contract", () => { + it("is a non-empty system notice carrying the task fan-out contract", () => { expect(WORKFLOW_NOTICE.length).toBeGreaterThan(0); expect(WORKFLOW_NOTICE).toContain("**workflowz** keyword"); - expect(WORKFLOW_NOTICE).toContain("parallel("); + expect(WORKFLOW_NOTICE).toContain("Use the `task` tool for batched fan-out"); + expect(WORKFLOW_NOTICE).toContain("tasks[]"); }); }); diff --git a/packages/coding-agent/test/system-prompt-inventory.test.ts b/packages/coding-agent/test/system-prompt-inventory.test.ts index 82b51b791..cf06b0cd5 100644 --- a/packages/coding-agent/test/system-prompt-inventory.test.ts +++ b/packages/coding-agent/test/system-prompt-inventory.test.ts @@ -164,6 +164,9 @@ describe("system prompt tool inventory", () => { expect(bash?.description).toContain("purpose-built tool"); expect(bash?.description).not.toContain("eval` cell"); expect(bash?.description).not.toContain("use `eval` cells"); + expect(bash?.description).not.toContain("`grep` tool"); + expect(bash?.description).not.toContain("`ls` → `read`"); + expect(bash?.description).not.toContain("`find` → the `glob` tool"); const { systemPrompt } = await buildSystemPrompt({ cwd: tempDir, From b9a12a815f474215eb77f10b7ff9c98485b272d9 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 17:27:35 -0700 Subject: [PATCH 40/91] fix(tui): keep autocomplete row fix behavior-preserving Signed-off-by: Christian Stewart --- packages/coding-agent/CHANGELOG.md | 3 --- packages/coding-agent/src/config/settings-schema.ts | 3 ++- packages/tui/CHANGELOG.md | 2 +- packages/tui/src/components/editor.ts | 2 +- packages/tui/test/editor.test.ts | 1 + 5 files changed, 5 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a939e8b83..65cdbf02c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,9 +2,6 @@ ## [Unreleased] -### Changed - -- Increased the default autocomplete dropdown from 5 to 10 visible items so slash command menus are less cramped. ## [16.3.8] - 2026-07-05 diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index da512d952..00af4ed69 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -1493,7 +1493,7 @@ export const SETTINGS_SCHEMA = { autocompleteMaxVisible: { type: "number", - default: 10, + default: 5, ui: { tab: "interaction", group: "Input", @@ -1502,6 +1502,7 @@ export const SETTINGS_SCHEMA = { options: [ { value: "3", label: "3 items" }, { value: "5", label: "5 items" }, + { value: "7", label: "7 items" }, { value: "10", label: "10 items" }, { value: "15", label: "15 items" }, { value: "20", label: "20 items" }, diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 8c01a59cd..1c864ff3e 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Kept slash command autocomplete rows compact by truncating descriptions instead of wrapping them into multi-line blocks, and raised the editor's default autocomplete window to 10 items. +- Kept slash command autocomplete rows compact by truncating descriptions instead of wrapping them into multi-line blocks. ## [16.3.7] - 2026-07-05 diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 4d88f266f..10e23a1ab 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -422,7 +422,7 @@ export class Editor implements Component, Focusable { #autocompleteState: "regular" | "force" | null = null; #autocompletePrefix: string = ""; #autocompleteRequestId: number = 0; - #autocompleteMaxVisible: number = 10; + #autocompleteMaxVisible: number = 5; onAutocompleteUpdate?: () => void; // Paste tracking for large pastes diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index 9240aebb0..da3f0f7d3 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -311,6 +311,7 @@ describe("Editor component", () => { it("renders slash-command suggestions as compact item rows", async () => { const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteMaxVisible(10); const longDescription = "Plan and execute non-trivial architectural improvements to the codebase without turning each slash command into a multi-line block."; editor.setAutocompleteProvider( From 01451314322cfdd1a2f0b9681e4516b6dadaa445 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 17:28:30 -0700 Subject: [PATCH 41/91] test(bash): defend zero-timeout executor deadline Signed-off-by: Christian Stewart --- packages/coding-agent/CHANGELOG.md | 4 ++++ packages/coding-agent/test/bash-executor.test.ts | 2 +- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 702ba1d2e..99befc6ee 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `bash` tool `timeout: 0` so it disables the command deadline instead of falling back to the minimum timeout. + ## [16.3.8] - 2026-07-05 ### Fixed diff --git a/packages/coding-agent/test/bash-executor.test.ts b/packages/coding-agent/test/bash-executor.test.ts index 2e32fd4ae..b11a15f96 100644 --- a/packages/coding-agent/test/bash-executor.test.ts +++ b/packages/coding-agent/test/bash-executor.test.ts @@ -406,7 +406,7 @@ exit 64 if (process.platform === "win32") { return; } - const result = await executeBash("sleep 0.1; echo done", { cwd: tempDir, timeout: 0 }); + const result = await executeBash("sleep 1.2; echo done", { cwd: tempDir, timeout: 0 }); expect(result.cancelled).toBe(false); expect(result.output.trim()).toBe("done"); }); From 58a3259abf05337b98c4c504ae487bab8ffa48de Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 17:30:46 -0700 Subject: [PATCH 42/91] test(prompt): stabilize local-date regressions Signed-off-by: Christian Stewart --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/session/agent-session.ts | 6 +- .../agent-session-tool-rebuild-skip.test.ts | 62 +++++------- .../test/system-prompt-model.test.ts | 99 ++++++++++++++----- 4 files changed, 109 insertions(+), 62 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 702ba1d2e..858e2c352 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed system prompt date rendering to use the host local calendar date instead of UTC. + ## [16.3.8] - 2026-07-05 ### Fixed diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 5e49cba05..2483752f6 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -725,6 +725,8 @@ export interface AgentSessionConfig { convertToLlm?: (messages: AgentMessage[]) => Message[] | Promise; /** System prompt builder that can consider tool availability. Returns ordered provider-facing blocks. */ rebuildSystemPrompt?: (toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>; + /** Local calendar date provider used by prompt-cache invalidation. Defaults to the host local date. */ + getLocalCalendarDate?: () => string; /** Rebuild the SSH tool from current capability discovery results. */ reloadSshTool?: () => Promise; requestedToolNames?: ReadonlySet; @@ -1716,6 +1718,7 @@ export class AgentSession { #rebuildSystemPrompt: | ((toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>) | undefined; + #getLocalCalendarDate: () => string; #getMcpServerInstructions: (() => Map | undefined) | undefined; #reloadSshTool: (() => Promise) | undefined; #disconnectOwnedMcpManager: (() => Promise) | undefined; @@ -2154,6 +2157,7 @@ export class AgentSession { }); this.#convertToLlm = config.convertToLlm ?? convertToLlm; this.#rebuildSystemPrompt = config.rebuildSystemPrompt; + this.#getLocalCalendarDate = config.getLocalCalendarDate ?? formatLocalCalendarDate; this.#getMcpServerInstructions = config.getMcpServerInstructions; this.#reloadSshTool = config.reloadSshTool; this.#disconnectOwnedMcpManager = config.disconnectOwnedMcpManager; @@ -6518,7 +6522,7 @@ export class AgentSession { entries.sort(); instructionsSegment = entries.join("\u0006"); } - const date = formatLocalCalendarDate(); + const date = this.#getLocalCalendarDate(); return `${nameSegment}\u0003${descriptionSegment}\u0005${registrySegment}\u0007${instructionsSegment}|${date}`; } diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 53f795f2c..c1ae17fee 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, it, setSystemTime } from "bun:test"; +import { afterEach, describe, expect, it } from "bun:test"; import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; import type { Model } from "@oh-my-pi/pi-ai"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; @@ -68,6 +68,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { interface NewSessionOptions { mcpDiscoveryEnabled?: boolean; getMcpServerInstructions?: () => Map | undefined; + getLocalCalendarDate?: () => string; } function newSession( @@ -101,6 +102,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { }), mcpDiscoveryEnabled: options.mcpDiscoveryEnabled, getMcpServerInstructions: options.getMcpServerInstructions, + getLocalCalendarDate: options.getLocalCalendarDate, }); sessions.push(session); return { session }; @@ -389,47 +391,37 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { expect(rebuildCount).toBe(baseline + 1); }); it("rebuilds when the local calendar date rolls over between tool-stable MCP refreshes", async () => { - // `buildSystemPrompt` injects today's date into the prompt body. A session - // spanning local midnight must not serve yesterday's date after an MCP - // reconnect that happens to bring an identical tool set, even when UTC has - // not rolled over. - const originalTimezone = process.env.TZ; - process.env.TZ = "America/Los_Angeles"; - setSystemTime(new Date("2026-07-01T06:59:58Z")); - try { - let rebuildCount = 0; - const { session } = newSession(async toolNames => { + // `buildSystemPrompt` injects today's local date into the prompt body. The + // signature reads the same date provider so a session spanning local midnight + // must rebuild after an MCP reconnect with an otherwise identical tool set. + let currentDate = "2026-06-30"; + let rebuildCount = 0; + const { session } = newSession( + async toolNames => { rebuildCount++; return `tools:${toolNames.join(",")}`; - }); - const tool = createMcpCustomTool("mcp__nucleus_search", "nucleus", "search", "Search"); + }, + { getLocalCalendarDate: () => currentDate }, + ); + const tool = createMcpCustomTool("mcp__nucleus_search", "nucleus", "search", "Search"); - // First refresh: no signature yet, must rebuild. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(1); + // First refresh: no signature yet, must rebuild. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(1); - // Same tools, same local day: signature matches, skip. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(1); + // Same tools, same local day: signature matches, skip. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(1); - // Advance past local midnight while the UTC date remains 2026-07-01. - setSystemTime(new Date("2026-07-01T07:00:01Z")); + currentDate = "2026-07-01"; - // Same tools, new local calendar day: date segment changed, must rebuild. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(2); + // Same tools, new local calendar day: date segment changed, must rebuild. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(2); - // Same tools, same new local day: skip again. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(2); - } finally { - setSystemTime(); - if (originalTimezone === undefined) { - delete process.env.TZ; - } else { - process.env.TZ = originalTimezone; - } - } + // Same tools, same new local day: skip again. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(2); }); it("does not rebuild when MCP server instructions change only beyond the 4000-char truncation boundary", async () => { // `rebuildSystemPrompt` (sdk.ts) truncates each server instruction to 4000 chars diff --git a/packages/coding-agent/test/system-prompt-model.test.ts b/packages/coding-agent/test/system-prompt-model.test.ts index 3f253052e..276e103ac 100644 --- a/packages/coding-agent/test/system-prompt-model.test.ts +++ b/packages/coding-agent/test/system-prompt-model.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it, setSystemTime } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -21,6 +21,69 @@ const EMPTY_TREE = { agentsMdFiles: [], }; +async function expectPromptDateFromStartupTimezone(options: { + tempDir: string; + tempHomeDir: string; + timeZone: string; + now: string; + expectedDate: string; + rejectedDate: string; +}): Promise { + const scenarioPath = path.join(options.tempDir, "prompt-date-timezone.test.ts"); + await Bun.write( + scenarioPath, + `import { expect, it, setSystemTime } from "bun:test"; +import { buildSystemPrompt } from ${JSON.stringify(path.resolve(import.meta.dir, "../src/system-prompt.ts"))}; + +it("renders the prompt date in the startup timezone", async () => { + setSystemTime(new Date(process.env.OMP_TEST_NOW!)); + try { + const { systemPrompt } = await buildSystemPrompt({ + cwd: process.cwd(), + contextFiles: [], + skills: [], + rules: [], + toolNames: [], + workspaceTree: { + rootPath: process.cwd(), + rendered: "", + truncated: false, + totalLines: 0, + agentsMdFiles: [], + }, + activeRepoContext: null, + }); + const rendered = systemPrompt.join("\\n\\n"); + expect(rendered).toContain(\`Today is \${process.env.OMP_EXPECTED_DATE}\`); + expect(rendered).not.toContain(\`Today is \${process.env.OMP_REJECTED_DATE}\`); + } finally { + setSystemTime(); + } +}); +`, + ); + const child = Bun.spawn([process.execPath, "test", scenarioPath], { + cwd: options.tempDir, + env: { + ...process.env, + HOME: options.tempHomeDir, + TZ: options.timeZone, + OMP_TEST_NOW: options.now, + OMP_EXPECTED_DATE: options.expectedDate, + OMP_REJECTED_DATE: options.rejectedDate, + }, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(child.stdout).text(), + new Response(child.stderr).text(), + child.exited, + ]); + expect(`${stdout}\n${stderr}`).toContain("1 pass"); + expect(exitCode).toBe(0); +} + describe("system prompt model identifier", () => { let tempDir = ""; let tempHomeDir = ""; @@ -49,31 +112,15 @@ describe("system prompt model identifier", () => { expect(systemPrompt.join("\n\n")).toContain("Model: anthropic/claude-opus-4"); }); - it("renders the prompt date from the local timezone rather than UTC", async () => { - const originalTimezone = process.env.TZ; - process.env.TZ = "America/Los_Angeles"; - setSystemTime(new Date("2026-07-01T03:15:00Z")); - try { - const { systemPrompt } = await buildSystemPrompt({ - cwd: tempDir, - contextFiles: [], - skills: [], - rules: [], - toolNames: [], - workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, - }); - const rendered = systemPrompt.join("\n\n"); - - expect(rendered).toContain("Today is 2026-06-30"); - expect(rendered).not.toContain("Today is 2026-07-01"); - } finally { - setSystemTime(); - if (originalTimezone === undefined) { - delete process.env.TZ; - } else { - process.env.TZ = originalTimezone; - } - } + it("renders the prompt date from the startup local timezone rather than UTC", async () => { + await expectPromptDateFromStartupTimezone({ + tempDir, + tempHomeDir, + timeZone: "America/Los_Angeles", + now: "2026-07-01T03:15:00Z", + expectedDate: "2026-06-30", + rejectedDate: "2026-07-01", + }); }); it("omits the model line when no model is provided", async () => { From 25c94fadba5a0e0cc12f44265b76bbb8962053f3 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 17:34:17 -0700 Subject: [PATCH 43/91] fix(prompting): match workflowz task schema Signed-off-by: Christian Stewart --- packages/coding-agent/src/modes/workflow.ts | 22 +++++++----- .../src/prompts/system/workflow-notice.md | 34 +++++++++++++++---- .../coding-agent/src/session/agent-session.ts | 4 +-- .../test/agent-session-magic-keywords.test.ts | 16 +++++++++ .../coding-agent/test/modes/workflow.test.ts | 10 +++++- 5 files changed, 69 insertions(+), 17 deletions(-) diff --git a/packages/coding-agent/src/modes/workflow.ts b/packages/coding-agent/src/modes/workflow.ts index ab7ae17fa..e8d4832cc 100644 --- a/packages/coding-agent/src/modes/workflow.ts +++ b/packages/coding-agent/src/modes/workflow.ts @@ -1,4 +1,5 @@ -import workflowNotice from "../prompts/system/workflow-notice.md" with { type: "text" }; +import { prompt } from "@oh-my-pi/pi-utils"; +import workflowNoticeTemplate from "../prompts/system/workflow-notice.md" with { type: "text" }; import { createGradientHighlighter, type KeywordHighlighter } from "./gradient-highlight"; import { keywordInProse } from "./markdown-prose"; @@ -7,18 +8,23 @@ import { keywordInProse } from "./markdown-prose"; * * Typing the standalone word in the input editor paints it with a warm * amber→green gradient ({@link highlightWorkflow}); submitting a message that - * mentions it appends a hidden {@link WORKFLOW_NOTICE} that steers the model to - * author a deterministic multi-subagent workflow in eval cells (agent/parallel/ - * pipeline). Matching is whitespace-delimited and case-sensitive (lowercase - * only) — "workflowz" triggers, but "workflowzed", "Workflowz", and - * "workflowz.ts" never do. + * mentions it appends a hidden workflow notice that steers the model to author + * a deterministic multi-subagent workflow through the active task schema. + * Matching is whitespace-delimited and case-sensitive (lowercase only) — + * "workflowz" triggers, but "workflowzed", "Workflowz", and "workflowz.ts" + * never do. */ // Detection: lowercase keyword flanked by whitespace or a string edge. Non-global so `.test` stays stateless. const WORKFLOW_WORD = /(? -The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Use the `task` tool for batched fan-out — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. +The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Use the `task` tool {{#if taskBatch}}for batched fan-out{{else}}once per independent subagent{{/if}} — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline first (list the files, scope the diff, find the call sites) to discover the work list, then fan out over it. Common shapes: @@ -11,6 +11,7 @@ Worth it when the task benefits from decomposition + parallel coverage, or from +{{#if taskBatch}} Call `task` once per independent fan-out batch. Put shared background in `context`, and put each independent work item in `tasks[]`. Do not emulate batching with shell loops or eval helper APIs. `context` must carry the shared contract: @@ -31,13 +32,24 @@ Each task assignment must be self-contained: # Acceptance Observable result, return packet, and local verification. Subagents skip formatters, linters, and project-wide tests; the parent runs shared proof once. +{{else}} +Call `task` once per independent subagent. Put the full shared background and the leaf work in that call's `assignment`. Do not pass `context` or `tasks[]`: the flat task schema rejects them when batch calls are disabled. -Use specific roles (`Storage Reviewer`, `CLI Migrator`, `Security Skeptic`) rather than generic workers. Dispatch at most one layer unless the user grants a named second-layer purpose. - +Each assignment must be self-contained: + + # Target + Exact files, symbols, subsystem, or evidence surface; explicit non-goals. + # Change + Shared background plus what to inspect or modify, step by step, including APIs and patterns to reuse. + # Acceptance + Observable result, return packet, and local verification. Subagents skip formatters, + linters, and project-wide tests; the parent runs shared proof once. +{{/if}} -Decompose first, then batch the independent leaves: +Decompose first, then {{#if taskBatch}}batch the independent leaves{{else}}issue one independent task call per leaf in the same turn{{/if}}: +{{#if taskBatch}} task( context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...", tasks: [ @@ -45,8 +57,18 @@ Decompose first, then batch the independent leaves: { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." }, ] ) +{{else}} + task( + role: "Auth Storage Reviewer", + assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only..." + ) + task( + role: "Prompt Contract Reviewer", + assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only..." + ) +{{/if}} -Prefer one wide batch over serial subagent calls when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing. +{{#if taskBatch}}Prefer one wide batch over serial subagent calls when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing.{{else}}Prefer issuing all independent task calls in one assistant turn over serial dispatch when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing.{{/if}} @@ -59,7 +81,7 @@ Prefer one wide batch over serial subagent calls when work items do not share fi - Capture multi-phase workflow state in the visible todo system when available. -- Batch independent subagents in one `task` call. +{{#if taskBatch}}- Batch independent subagents in one `task` call.{{else}}- Dispatch independent subagents as separate `task` calls in the same turn.{{/if}} - Give every subagent a narrow target, explicit non-goals, and a concrete return packet. - After fan-out returns, read the artifacts, patch or decide, and run the shared gate. - Keep going until the task is closed — returned fan-out is a step, not a stopping point. diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index da4492fc4..5b36e37a6 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -237,7 +237,7 @@ import { theme } from "../modes/theme/theme"; import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; import { computeNonMessageBreakdown, computeNonMessageTokens } from "../modes/utils/context-usage"; -import { containsWorkflow, WORKFLOW_NOTICE } from "../modes/workflow"; +import { containsWorkflow, renderWorkflowNotice } from "../modes/workflow"; import { createPlanReadMatcher } from "../plan-mode/plan-protection"; import type { PlanModeState } from "../plan-mode/state"; import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" }; @@ -7338,7 +7338,7 @@ export class AgentSession { keywordNotices.push({ role: "custom", customType: "workflow-notice", - content: WORKFLOW_NOTICE, + content: renderWorkflowNotice({ taskBatch: this.settings.get("task.batch") }), display: false, attribution: "user", timestamp, diff --git a/packages/coding-agent/test/agent-session-magic-keywords.test.ts b/packages/coding-agent/test/agent-session-magic-keywords.test.ts index d0e09a7ae..4ac59444d 100644 --- a/packages/coding-agent/test/agent-session-magic-keywords.test.ts +++ b/packages/coding-agent/test/agent-session-magic-keywords.test.ts @@ -103,6 +103,22 @@ describe("AgentSession magic keyword settings", () => { ]); }); + it("renders workflowz notice for the active task schema", async () => { + const created = await createMagicKeywordSession(root); + session = created.session; + authStorage = created.authStorage; + created.settings.set("task.batch", false); + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + + await session.prompt("please workflowz this"); + + const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ content?: string; customType?: string }>; + const notice = promptMessages.find(message => message.customType === "workflow-notice")?.content ?? ""; + expect(notice).toContain("once per independent subagent"); + expect(notice).toContain("Do not pass `context` or `tasks[]`"); + expect(notice).not.toContain("Call `task` once per independent fan-out batch"); + }); + it("does not use a disabled ultrathink keyword to force auto thinking", async () => { const created = await createMagicKeywordSession(root); session = created.session; diff --git a/packages/coding-agent/test/modes/workflow.test.ts b/packages/coding-agent/test/modes/workflow.test.ts index a9e719b5b..32b01f31b 100644 --- a/packages/coding-agent/test/modes/workflow.test.ts +++ b/packages/coding-agent/test/modes/workflow.test.ts @@ -1,6 +1,6 @@ import { beforeAll, describe, expect, it } from "bun:test"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { containsWorkflow, highlightWorkflow, WORKFLOW_NOTICE } from "@oh-my-pi/pi-coding-agent/modes/workflow"; +import { containsWorkflow, highlightWorkflow, renderWorkflowNotice, WORKFLOW_NOTICE } from "@oh-my-pi/pi-coding-agent/modes/workflow"; beforeAll(() => { // highlightWorkflow reads the global theme's color mode. @@ -54,4 +54,12 @@ describe("workflow notice", () => { expect(WORKFLOW_NOTICE).toContain("Use the `task` tool for batched fan-out"); expect(WORKFLOW_NOTICE).toContain("tasks[]"); }); + + it("renders flat task-call guidance when task.batch is disabled", () => { + const notice = renderWorkflowNotice({ taskBatch: false }); + expect(notice).toContain("once per independent subagent"); + expect(notice).toContain("Do not pass `context` or `tasks[]`"); + expect(notice).toContain("one independent task call per leaf"); + expect(notice).not.toContain("Call `task` once per independent fan-out batch"); + }); }); From 96850a1642c0f8b0f9cb92d9b67e06f27d2ef067 Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 18:04:24 -0700 Subject: [PATCH 44/91] fix(prompting): keep eval-disabled tool state coherent Signed-off-by: Christian Stewart --- packages/coding-agent/src/prompts/tools/bash.md | 2 +- packages/coding-agent/src/tools/index.ts | 7 ++----- .../coding-agent/test/system-prompt-inventory.test.ts | 1 + packages/coding-agent/test/tools/index.test.ts | 9 +++++++++ 4 files changed, 13 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 9af269ac3..f95e22d8d 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -31,7 +31,7 @@ Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a f - `;` only when later commands should run despite earlier failures - Multiple bash calls per message run concurrently. NEVER split order-dependent commands across parallel calls — chain with `&&` in one call. - Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths -- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly. +{{#if hasEval}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly.{{else}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Use a checked-in script, purpose-built tool, or single command that owns the output shape.{{/if}} {{#if asyncEnabled}} - `async: true` for long-running commands when you don't need immediate output: returns a background job ID; result delivered as a follow-up. {{/if}} diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 621c80f86..83e39a517 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -650,14 +650,11 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P ]; const activeToolNames = new Set(baseEntries.map(([name]) => name)); - const toolFactorySession: ToolSession = { - ...session, - isToolActive: name => activeToolNames.has(name), - }; + session.isToolActive = name => activeToolNames.has(name); const baseResults = await Promise.all( baseEntries.map(async ([name, factory]) => { - const tool = await logger.time(`createTools:${name}`, factory as ToolFactory, toolFactorySession); + const tool = await logger.time(`createTools:${name}`, factory as ToolFactory, session); return tool ? wrapToolWithMetaNotice(tool) : null; }), ); diff --git a/packages/coding-agent/test/system-prompt-inventory.test.ts b/packages/coding-agent/test/system-prompt-inventory.test.ts index cf06b0cd5..a142f978a 100644 --- a/packages/coding-agent/test/system-prompt-inventory.test.ts +++ b/packages/coding-agent/test/system-prompt-inventory.test.ts @@ -164,6 +164,7 @@ describe("system prompt tool inventory", () => { expect(bash?.description).toContain("purpose-built tool"); expect(bash?.description).not.toContain("eval` cell"); expect(bash?.description).not.toContain("use `eval` cells"); + expect(bash?.description).not.toContain("Prefer `eval`"); expect(bash?.description).not.toContain("`grep` tool"); expect(bash?.description).not.toContain("`ls` → `read`"); expect(bash?.description).not.toContain("`find` → the `glob` tool"); diff --git a/packages/coding-agent/test/tools/index.test.ts b/packages/coding-agent/test/tools/index.test.ts index eae16b79c..8978dbfb3 100644 --- a/packages/coding-agent/test/tools/index.test.ts +++ b/packages/coding-agent/test/tools/index.test.ts @@ -264,6 +264,15 @@ describe("createTools", () => { expect(names).toEqual(["read", "goal", "resolve"]); }); + it("records active tools on the original session object", async () => { + const session = createTestSession(); + + await createTools(session, ["bash"]); + + expect(session.isToolActive?.("bash")).toBe(true); + expect(session.isToolActive?.("read")).toBe(false); + }); + it("includes search_tool_bm25 when MCP tool discovery is enabled and executable", async () => { const session = createTestSession({ settings: createSettingsWithOverrides({ From 1936d4f25003ea03551c98fd08d05916633536bf Mon Sep 17 00:00:00 2001 From: Christian Stewart Date: Sun, 5 Jul 2026 18:15:22 -0700 Subject: [PATCH 45/91] fix(prompting): refresh bash guidance on tool changes Signed-off-by: Christian Stewart --- packages/coding-agent/src/sdk.ts | 11 +++++ .../coding-agent/src/session/agent-session.ts | 6 +++ packages/coding-agent/src/tools/bash.ts | 29 ++++++------ packages/coding-agent/src/tools/index.ts | 8 +++- .../agent-session-tool-rebuild-skip.test.ts | 46 +++++++++++++++++++ .../coding-agent/test/tools/index.test.ts | 22 +++++++++ 6 files changed, 107 insertions(+), 15 deletions(-) diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index bca73c40e..3321e78ac 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1525,10 +1525,19 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // entries capture it at fetch time and are dropped at injection if a newer // mutation (any tool) bumped it in the meantime. const fileMutationVersions = new Map(); + const activeToolNames = new Set(); + const setActiveToolNames = (names: Iterable): void => { + activeToolNames.clear(); + for (const name of names) { + activeToolNames.add(name); + } + }; const toolSession: ToolSession = { get cwd() { return sessionManager.getCwd(); }, + isToolActive: name => activeToolNames.has(name), + setActiveToolNames, hasUI: options.hasUI ?? false, enableLsp, get hasEditTool() { @@ -2540,6 +2549,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }); hasRegistered = true; + setActiveToolNames(initialToolNames); const { systemPrompt } = await logger.time( "buildSystemPrompt", rebuildSystemPrompt, @@ -2834,6 +2844,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} rebuildSystemPrompt, reloadSshTool, requestedToolNames: requestedToolNameSet, + setActiveToolNames, getMcpServerInstructions: mcpManager ? () => { const raw = mcpManager.getServerInstructions(); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 5b36e37a6..8be2067d4 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -686,6 +686,8 @@ export interface AgentSessionConfig { toolRegistry?: Map; /** Tool names whose current registry entry is still the built-in implementation. */ builtInToolNames?: Iterable; + /** Update tool-session predicates that render guidance from the live active tool set. */ + setActiveToolNames?: (names: Iterable) => void; /** Current session pre-LLM message transform pipeline */ transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise; /** @@ -1717,6 +1719,7 @@ export class AgentSession { | undefined; #getMcpServerInstructions: (() => Map | undefined) | undefined; #reloadSshTool: (() => Promise) | undefined; + #setActiveToolNames: ((names: Iterable) => void) | undefined; #disconnectOwnedMcpManager: (() => Promise) | undefined; #requestedToolNames: ReadonlySet | undefined; #baseSystemPrompt: string[]; @@ -2155,6 +2158,7 @@ export class AgentSession { this.#rebuildSystemPrompt = config.rebuildSystemPrompt; this.#getMcpServerInstructions = config.getMcpServerInstructions; this.#reloadSshTool = config.reloadSshTool; + this.#setActiveToolNames = config.setActiveToolNames; this.#disconnectOwnedMcpManager = config.disconnectOwnedMcpManager; this.#baseSystemPrompt = this.agent.state.systemPrompt; this.#promptModelKey = this.#currentPromptModelKey(); @@ -6290,6 +6294,7 @@ export class AgentSession { ), ); } + this.#setActiveToolNames?.(validToolNames); const activeNameSet = new Set(validToolNames); for (const name of Array.from(this.#selectedDiscoveredToolNames)) { if (!activeNameSet.has(name) || isMCPToolName(name) || !this.#toolRegistry.has(name)) { @@ -6395,6 +6400,7 @@ export class AgentSession { async refreshBaseSystemPrompt(): Promise { if (!this.#rebuildSystemPrompt) return; const activeToolNames = this.getActiveToolNames(); + this.#setActiveToolNames?.(activeToolNames); const built = await this.#rebuildSystemPrompt(activeToolNames, this.#toolRegistry); this.#baseSystemPrompt = built.systemPrompt; this.#baseSystemPromptBeforeMemoryPromotion = undefined; diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index 9feef6a76..ea9f4a031 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -376,7 +376,21 @@ export class BashTool implements AgentTool this.session.isToolActive?.(name) ?? fallback; + return prompt.render(bashDescription, { + asyncEnabled: this.#asyncEnabled, + autoBackgroundEnabled: this.#autoBackgroundEnabled, + autoBackgroundThresholdSeconds: Math.max(0, Math.floor(this.#autoBackgroundThresholdMs / 1000)), + hasAstGrep: isToolActive("ast_grep", this.session.settings.get("astGrep.enabled")), + hasAstEdit: isToolActive("ast_edit", this.session.settings.get("astEdit.enabled")), + hasGrep: isToolActive("grep", this.session.settings.get("grep.enabled")), + hasGlob: isToolActive("glob", this.session.settings.get("glob.enabled")), + hasRead: isToolActive("read", true), + hasEval: isToolActive("eval", evalBackends.python || evalBackends.js || evalBackends.ruby || evalBackends.julia), + }); + } readonly parameters: BashToolSchema; // Non-pty calls run alongside each other (the executor isolates overlapping // runs on the same shell session); pty takes over the terminal UI and must @@ -398,19 +412,6 @@ export class BashTool implements AgentTool this.session.isToolActive?.(name) ?? fallback; - this.description = prompt.render(bashDescription, { - asyncEnabled: this.#asyncEnabled, - autoBackgroundEnabled: this.#autoBackgroundEnabled, - autoBackgroundThresholdSeconds: Math.max(0, Math.floor(this.#autoBackgroundThresholdMs / 1000)), - hasAstGrep: isToolActive("ast_grep", this.session.settings.get("astGrep.enabled")), - hasAstEdit: isToolActive("ast_edit", this.session.settings.get("astEdit.enabled")), - hasGrep: isToolActive("grep", this.session.settings.get("grep.enabled")), - hasGlob: isToolActive("glob", this.session.settings.get("glob.enabled")), - hasRead: isToolActive("read", true), - hasEval: isToolActive("eval", evalBackends.python || evalBackends.js || evalBackends.ruby || evalBackends.julia), - }); } #formatResultOutput(result: BashResult | BashInteractiveResult): string { diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 83e39a517..8ea30c85d 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -226,6 +226,8 @@ export interface ToolSession { getToolByName?: (name: string) => AgentTool | undefined; /** Return whether a built-in tool is active in this turn's tool set. */ isToolActive?: (name: string) => boolean; + /** Update the active built-in tool predicate when a session changes tools mid-run. */ + setActiveToolNames?: (names: Iterable) => void; /** Agent registry for IRC routing across live sessions. */ agentRegistry?: AgentRegistry; /** Get artifacts directory for artifact:// URLs */ @@ -650,7 +652,11 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P ]; const activeToolNames = new Set(baseEntries.map(([name]) => name)); - session.isToolActive = name => activeToolNames.has(name); + if (session.setActiveToolNames) { + session.setActiveToolNames(activeToolNames); + } else { + session.isToolActive = name => activeToolNames.has(name); + } const baseResults = await Promise.all( baseEntries.map(async ([name, factory]) => { diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 6cefd9c5b..221720e06 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -176,6 +176,52 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { expect(rebuildCount).toBe(baseline + 2); }); + it("updates live active-tool predicates before rebuilding the prompt", async () => { + const activeToolNames = new Set(["read", "bash", "grep"]); + const readTool = createBasicTool("read", "Read"); + const bashTool = createBasicTool("bash", "Bash"); + const grepTool = createBasicTool("grep", "Grep"); + Object.defineProperty(bashTool, "description", { + get: () => (activeToolNames.has("grep") ? "bash sees grep" : "bash hides grep"), + enumerable: true, + configurable: true, + }); + const toolRegistry = new Map([ + [readTool.name, readTool], + [bashTool.name, bashTool], + [grepTool.name, grepTool], + ]); + const agent = new Agent({ + initialState: { + model: createModel(), + systemPrompt: ["initial"], + tools: [readTool, bashTool, grepTool], + messages: [], + }, + }); + const session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry: {} as never, + toolRegistry, + setActiveToolNames: names => { + activeToolNames.clear(); + for (const name of names) { + activeToolNames.add(name); + } + }, + rebuildSystemPrompt: async (_toolNames, tools) => ({ + systemPrompt: [tools.get("bash")?.description ?? "missing bash"], + }), + }); + sessions.push(session); + + await session.setActiveToolsByName(["read", "bash"]); + + expect(agent.state.systemPrompt).toEqual(["bash hides grep"]); + }); + it("does not skip when refreshBaseSystemPrompt is called explicitly", async () => { let rebuildCount = 0; const { session } = newSession(async toolNames => { diff --git a/packages/coding-agent/test/tools/index.test.ts b/packages/coding-agent/test/tools/index.test.ts index 8978dbfb3..845a285b1 100644 --- a/packages/coding-agent/test/tools/index.test.ts +++ b/packages/coding-agent/test/tools/index.test.ts @@ -273,6 +273,28 @@ describe("createTools", () => { expect(session.isToolActive?.("read")).toBe(false); }); + it("renders bash guidance from the live active tool predicate", async () => { + const activeToolNames = new Set(); + const session = createTestSession({ + isToolActive: name => activeToolNames.has(name), + setActiveToolNames: names => { + activeToolNames.clear(); + for (const name of names) { + activeToolNames.add(name); + } + }, + }); + + const tools = await createTools(session, ["bash", "grep", "read", "glob"]); + const bash = tools.find(tool => tool.name === "bash"); + + expect(bash?.description).toContain("`grep` tool"); + session.setActiveToolNames?.(["bash"]); + expect(bash?.description).not.toContain("`grep` tool"); + expect(bash?.description).not.toContain("`ls` → `read`"); + expect(bash?.description).not.toContain("`find` → the `glob` tool"); + }); + it("includes search_tool_bm25 when MCP tool discovery is enabled and executable", async () => { const session = createTestSession({ settings: createSettingsWithOverrides({ From 2d180b885f4a7a0cba9ae59022a5400113d762ad Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 06:13:55 +0000 Subject: [PATCH 46/91] fix(coding-agent): guarded pasted shell prompts from python - Detected copied shell-prompt transcripts before the Python shortcut router. - Forwarded OMP terminal chrome pastes through normal prompt submission. - Added regression coverage for the #4678 transcript shape. Fixes #4678 --- packages/coding-agent/CHANGELOG.md | 4 +++ .../src/modes/controllers/input-controller.ts | 18 ++++++++++- .../input-controller-python-prefix.test.ts | 30 +++++++++++++++++++ 3 files changed, 51 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c5595a893..514e1491c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed pasted terminal transcripts beginning with a shell prompt (`$ ...`) being mistaken for local Python shortcuts instead of being submitted as normal prompts ([#4678](https://github.com/can1357/oh-my-pi/issues/4678)). + ## [16.3.9] - 2026-07-06 ### Added diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index ed16eb36c..9fd238744 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -86,6 +86,20 @@ function hasPasteText(value: unknown): value is PasteTarget { return typeof value === "object" && value !== null && typeof (value as PasteTarget).pasteText === "function"; } +const SHELL_PROMPT_COMMAND_RE = + /^(?:\.{0,2}\/|~\/|cd(?:\s|$)|sudo(?:\s|$)|git(?:\s|$)|bun(?:\s|$)|npm(?:\s|$)|pnpm(?:\s|$)|yarn(?:\s|$)|node(?:\s|$)|python\d*(?:\s|$)|cargo(?:\s|$)|go(?:\s|$)|make(?:\s|$)|docker(?:\s|$)|kubectl(?:\s|$))/; +const SHELL_PROMPT_OPERATOR_RE = /(?:^|\s)(?:&&|\|\||\||2>&1|[<>]{1,2})(?:\s|$)/; +const OMP_STATUS_LINE_RE = /^\s*in:\s+\d+\s+out:\s+\d+(?:\s+cache\s+\S+)?\s+t:\s+\S+\s+tok\/s:\s+\S+/m; + +function looksLikePastedShellPrompt(code: string): boolean { + const firstLine = code.split("\n", 1)[0]?.trimStart() ?? ""; + return ( + SHELL_PROMPT_COMMAND_RE.test(firstLine) || + SHELL_PROMPT_OPERATOR_RE.test(firstLine) || + OMP_STATUS_LINE_RE.test(code) + ); +} + function pythonCommandPrefixLength(trimmedText: string): 0 | 1 | 2 { if (trimmedText.charCodeAt(0) !== 36 /* $ */) return 0; if (trimmedText.charCodeAt(1) === 123 /* { */) return 0; @@ -100,8 +114,10 @@ function parsePythonCommandInput(text: string): { code: string; isExcluded: bool const trimmed = text.trimStart(); const prefixLength = pythonCommandPrefixLength(trimmed); if (prefixLength === 0) return undefined; + const code = trimmed.slice(prefixLength).trim(); + if (prefixLength === 1 && looksLikePastedShellPrompt(code)) return undefined; return { - code: trimmed.slice(prefixLength).trim(), + code, isExcluded: prefixLength === 2, }; } diff --git a/packages/coding-agent/test/input-controller-python-prefix.test.ts b/packages/coding-agent/test/input-controller-python-prefix.test.ts index 92cd133ca..6e48d1af4 100644 --- a/packages/coding-agent/test/input-controller-python-prefix.test.ts +++ b/packages/coding-agent/test/input-controller-python-prefix.test.ts @@ -109,6 +109,36 @@ describe("InputController Python prompt prefix", () => { ]); }); + it("submits pasted shell-prompt transcripts with OMP chrome as a normal prompt", async () => { + const transcript = + "$ cd ~/project && sudo ./build-and-push.sh o5.7 2>&1 | tail -4\n" + + " |\n" + + " in: 282 out: 152 cache 344K t: 3.3s tok/s: 351.9/s\n" + + " is this command stuck in limbo"; + const { ctx, editor, handlePythonCommand, onInputCallback, startPendingSubmission, submitted } = createContext(); + const controller = new InputController(ctx); + controller.setupEditorSubmitHandler(); + + await editor.onSubmit?.(transcript); + + expect(handlePythonCommand).not.toHaveBeenCalled(); + expect(startPendingSubmission).toHaveBeenCalledWith({ + text: transcript, + images: undefined, + imageLinks: undefined, + streamingBehavior: "steer", + }); + expect(onInputCallback).toHaveBeenCalledTimes(1); + expect(submitted).toEqual([ + { + text: transcript, + images: undefined, + imageLinks: undefined, + streamingBehavior: "steer", + }, + ]); + }); + it("keeps space-separated Python shortcuts available", async () => { const { ctx, editor, handlePythonCommand, onInputCallback } = createContext(); const controller = new InputController(ctx); From 41a29c83c89053e22acba696d8cd7a85e70bd3c3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 06:19:01 +0000 Subject: [PATCH 47/91] fix(providers): disabled strict tools on azure anthropic Azure Foundry Anthropic routes reject Anthropic structured-output strict tooling for Sonnet 5 utility requests. Detect Azure Anthropic hosts as strict-tool-incompatible, gate the structured-output beta when strict tools are disabled, and cover the utility header plus tool-schema contracts. Fixes #4679 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/anthropic.ts | 18 +++- packages/ai/test/issue-4679-repro.test.ts | 104 ++++++++++++++++++++++ packages/catalog/CHANGELOG.md | 4 + packages/catalog/src/compat/anthropic.ts | 13 ++- 5 files changed, 132 insertions(+), 11 deletions(-) create mode 100644 packages/ai/test/issue-4679-repro.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d49636d82..b6c527af1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Azure Foundry Anthropic utility requests to omit the structured-output beta whenever strict tools are disabled, preventing `structured_outputs not supported in your workspace` failures for Sonnet 5 compaction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index b4949d838..7bef77e49 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -127,12 +127,13 @@ export function buildBetaHeader(baseBetas: readonly string[], extraBetas: readon const midConversationSystemBeta = "mid-conversation-system-2026-04-07"; const contextManagementBeta = "context-management-2025-06-27"; +const structuredOutputsBeta = "structured-outputs-2025-12-15"; const claudeCodeUtilityBetaDefaults = [ "oauth-2025-04-20", "interleaved-thinking-2025-05-14", contextManagementBeta, "prompt-caching-scope-2026-01-05", - "structured-outputs-2025-12-15", + structuredOutputsBeta, ] as const; const claudeCodeAgentBetaDefaults = [ "claude-code-20250219", @@ -159,10 +160,12 @@ function buildClaudeCodeBetas( agentRequest: boolean, thinkingRequest: boolean, redactThinking: boolean, + disableStrictTools = false, ): readonly string[] { - if (!agentRequest && !redactThinking) return claudeCodeUtilityBetaDefaults; + if (!agentRequest && !redactThinking && !disableStrictTools) return claudeCodeUtilityBetaDefaults; const betas: string[] = []; for (const beta of agentRequest ? claudeCodeAgentBetaDefaults : claudeCodeUtilityBetaDefaults) { + if (disableStrictTools && beta === structuredOutputsBeta) continue; betas.push(beta); // Match CC's header order: redact-thinking immediately follows interleaved-thinking. if (redactThinking && beta === interleavedThinkingBeta) betas.push(redactThinkingBeta); @@ -1098,6 +1101,7 @@ export type AnthropicClientOptionsArgs = { hasTools?: boolean; thinkingEnabled?: boolean; thinkingDisplay?: AnthropicThinkingDisplay; + disableStrictTools?: boolean; fetch?: FetchImpl; claudeCodeSessionId?: string; }; @@ -1833,6 +1837,7 @@ const streamAnthropicOnce = ( thinkingDisplay: options?.thinkingDisplay, fetch: options?.fetch, claudeCodeSessionId: options?.sessionId ?? extractClaudeMetadataSessionId(options?.metadata?.user_id), + disableStrictTools, }); client = created.client; isOAuthToken = created.isOAuthToken; @@ -2648,8 +2653,10 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A thinkingDisplay, isOAuth, claudeCodeSessionId, + disableStrictTools: disableStrictToolsOverride, } = args; const compat = model.compat; + const disableStrictTools = disableStrictToolsOverride ?? compat.disableStrictTools; const needsInterleavedBeta = interleavedThinking && !model.thinking?.supportsDisplay; const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming; const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey); @@ -2722,7 +2729,12 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A isCloudflareAiGateway: model.provider === "cloudflare-ai-gateway", claudeCodeSessionId, claudeCodeBetas: oauthToken - ? buildClaudeCodeBetas(hasTools || thinkingEnabled, thinkingEnabled, thinkingDisplay === "omitted") + ? buildClaudeCodeBetas( + hasTools || thinkingEnabled, + thinkingEnabled, + thinkingDisplay === "omitted", + disableStrictTools, + ) : [], }); diff --git a/packages/ai/test/issue-4679-repro.test.ts b/packages/ai/test/issue-4679-repro.test.ts new file mode 100644 index 000000000..2372bfc2d --- /dev/null +++ b/packages/ai/test/issue-4679-repro.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, it } from "bun:test"; +import { buildAnthropicClientOptions, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { Context, Model, ModelSpec, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const STRUCTURED_OUTPUTS_BETA = "structured-outputs-2025-12-15"; + +const bashTool: Tool = { + name: "bash", + description: "run a bash command", + parameters: { + type: "object", + properties: { command: { type: "string" } }, + required: ["command"], + } satisfies TJsonSchema, +}; + +const toolContext: Context = { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: 0 }], + tools: [bashTool], +}; + +function anthropicSpec(baseUrl: string): ModelSpec<"anthropic-messages"> { + return { + id: "claude-sonnet-5", + name: "Claude Sonnet 5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }; +} + +function buildOAuthUtilityBetaHeader(model: Model<"anthropic-messages">): string { + const options = buildAnthropicClientOptions({ + model, + apiKey: "oauth-token", + isOAuth: true, + hasTools: false, + thinkingEnabled: false, + }); + return options.defaultHeaders["anthropic-beta"] ?? ""; +} + +function abortedSignal(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +async function captureToolParams( + model: Model<"anthropic-messages">, +): Promise<{ tools?: Array<{ name: string; strict?: unknown }> }> { + const { promise, resolve } = Promise.withResolvers<{ tools?: Array<{ name: string; strict?: unknown }> }>(); + void streamAnthropic(model, toolContext, { + apiKey: "sk-ant-api-test", + isOAuth: false, + signal: abortedSignal(), + onPayload: payload => { + resolve(payload as { tools?: Array<{ name: string; strict?: unknown }> }); + return undefined; + }, + }); + return promise; +} + +describe("issue #4679 Azure Foundry Anthropic strict tools", () => { + it.each([ + ["inference", "https://example.inference.ai.azure.com/anthropic/v1"], + ["services", "https://example.services.ai.azure.com/anthropic/v1"], + ])("disables strict tools and omits structured-output beta for Azure Foundry %s routes", (_kind, baseUrl) => { + const model = buildModel(anthropicSpec(baseUrl)); + + expect(model.compat.disableStrictTools).toBe(true); + expect(buildOAuthUtilityBetaHeader(model)).not.toContain(STRUCTURED_OUTPUTS_BETA); + }); + + it("keeps structured-output beta on direct Anthropic OAuth utility headers", () => { + const model = buildModel(anthropicSpec("https://api.anthropic.com")); + + expect(model.compat.disableStrictTools).toBe(false); + expect(buildOAuthUtilityBetaHeader(model)).toContain(STRUCTURED_OUTPUTS_BETA); + }); + + it("omits strict tool schemas on Azure Foundry Anthropic requests without disabling direct Anthropic", async () => { + const azureParams = await captureToolParams( + buildModel(anthropicSpec("https://example.services.ai.azure.com/anthropic/v1")), + ); + const directParams = await captureToolParams(buildModel(anthropicSpec("https://api.anthropic.com"))); + + const azureBashTool = azureParams.tools?.find(tool => tool.name === "bash"); + const directBashTool = directParams.tools?.find(tool => tool.name === "bash"); + + expect(azureBashTool).toBeDefined(); + expect(azureBashTool?.strict).toBeUndefined(); + expect(directBashTool).toBeDefined(); + expect(directBashTool?.strict).toBe(true); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index ec9debe58..8f5be6e7b 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Detected Azure AI Inference / Foundry Anthropic routes as strict-tool-incompatible so resolved Anthropic compat disables strict tools before request construction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). + ## [16.3.9] - 2026-07-06 ### Fixed diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index 7bc1777c0..fafb6cac9 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -99,18 +99,15 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // (issue #4192). const isZenmux = modelMatchesHost(spec, "zenmux"); const requiresThinkingEnabled = modelMatchesHost(spec, "moonshotNative") && matchesKimiK27CodeFamily(spec); + const isVertex = isVertexAnthropicRoute(baseUrl); + const isBedrock = isBedrockAnthropicRoute(baseUrl); + const isAzure = isAzureAnthropicRoute(baseUrl); const signingEndpoint = - official || - isCopilot || - isZenmux || - isCloudflareAnthropicGateway(baseUrl) || - isVertexAnthropicRoute(baseUrl) || - isBedrockAnthropicRoute(baseUrl) || - isAzureAnthropicRoute(baseUrl); + official || isCopilot || isZenmux || isCloudflareAnthropicGateway(baseUrl) || isVertex || isBedrock || isAzure; const compat: ResolvedAnthropicCompat = { officialEndpoint: official, signingEndpoint, - disableStrictTools: false, + disableStrictTools: isAzure, disableAdaptiveThinking: false, supportsEagerToolInputStreaming: !isCopilot, // Long cache retention is only sent to the official API by default; From 47c172c70a02f7aa6324d4743cdf7ff1a2a7cbe2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 08:17:32 +0000 Subject: [PATCH 48/91] fix(coding-agent): handled json imports in plugin validator - Left JSON files imported with import attributes on Bun's native loader instead of registering them with the legacy source rewrite hook. - Added a regression test for loadLegacyPiModule loading a JSON import-attribute target. Fixes #4687 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../extensibility/plugins/legacy-pi-compat.ts | 8 ++++++-- .../extensibility/legacy-pi-inplace-load.test.ts | 16 ++++++++++++++++ 3 files changed, 26 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ec2a2142b..edf214d43 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed legacy plugin validation for extension graphs that import JSON with `with { type: "json" }`, leaving JSON files on Bun's native loader instead of parsing them as JavaScript ([#4687](https://github.com/can1357/oh-my-pi/issues/4687)). + ## [16.3.10] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index 1399be92e..20f55ad19 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -1059,7 +1059,8 @@ async function collectExtensionModules(entryRealPath: string): Promise { expect(mod.css).toBe(".x{color:red}"); }); + it("leaves JSON import-attribute targets on Bun's native loader", async () => { + const dir = await writePackage({ + "package.json": JSON.stringify({ name: "json-import-ext", version: "1.0.0" }), + "prices.json": JSON.stringify({ input: 0.15 }), + "index.ts": [ + 'import prices from "./prices.json" with { type: "json" };', + "export const inputPrice = prices.input;", + "export default function (pi) { void pi; }", + ].join("\n"), + }); + + const mod = (await loadLegacyPiModule(path.join(dir, "index.ts"))) as { inputPrice: number }; + + expect(mod.inputPrice).toBe(0.15); + }); + it("loads the extension's own node_modules deps natively while remapping legacy pi imports", async () => { const dir = await writePackage({ "package.json": JSON.stringify({ name: "dep-ext", version: "1.0.0" }), From 6ba13704b26dea3647b914014540f611bf2f9b92 Mon Sep 17 00:00:00 2001 From: ben Date: Mon, 6 Jul 2026 16:44:33 +0800 Subject: [PATCH 49/91] fix(ai): accept response.done terminal event --- packages/ai/CHANGELOG.md | 4 ++ packages/ai/src/providers/openai-responses.ts | 2 +- packages/ai/src/providers/openai-shared.ts | 26 +++++++--- .../test/openai-first-event-timeout.test.ts | 47 +++++++++++++++++++ 4 files changed, 72 insertions(+), 7 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 0bf512a7a..3e00e9368 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI Responses streams that end with `response.done` being misclassified as premature stream closures. + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 92e81e7fa..759fdbfdc 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -668,7 +668,7 @@ const streamOpenAIResponsesOnce = ( } // Detect premature stream closure: the HTTP stream ended without the - // provider sending `response.completed` or `response.incomplete`. + // provider sending a recognized terminal response event. // Custom/proxy providers may drop the connection mid-stream; without // this guard the incomplete output is silently surfaced as a successful // "stop". diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index f496cf713..1d40b0ac5 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -80,6 +80,7 @@ import { import type { ChatCompletionCreateParamsStreaming } from "./openai-chat-wire"; import type { InputItem } from "./openai-codex/request-transformer"; import type { + Response as OpenAIResponse, ResponseContentPartAddedEvent, ResponseCreateParamsStreaming, ResponseCustomToolCall, @@ -1798,13 +1799,25 @@ export function finalizeCustomToolCallInputDone(block: ResponsesToolCallBlock, i block.arguments = { input }; } +type OpenAIResponsesTerminalStreamEvent = + | Extract + | { type: "response.done"; response?: Partial }; + +function getOpenAIResponsesTerminalEvent(event: ResponseStreamEvent): OpenAIResponsesTerminalStreamEvent | undefined { + const type = (event as { type?: unknown }).type; + return type === "response.completed" || type === "response.incomplete" || type === "response.done" + ? (event as OpenAIResponsesTerminalStreamEvent) + : undefined; +} + export interface ProcessResponsesStreamOptions { onFirstToken?: () => void; onOutputItemDone?: (item: ResponseOutputItem) => void; /** - * Called when a terminal `response.completed` or `response.incomplete` event - * is successfully processed. Only invoked on the successful-completion path; - * thrown failure (`response.failed`) and cancellation paths never call this. + * Called when a terminal `response.completed`, `response.incomplete`, or + * `response.done` event is successfully processed. Only invoked on the + * successful-completion path; thrown failure (`response.failed`) and + * cancellation paths never call this. * Used by callers to detect premature stream closure (i.e. the stream ended * without a recognized terminal event). */ @@ -2039,6 +2052,7 @@ export async function processResponsesStream( let sawFirstToken = false; for await (const event of openaiStream) { + const terminalEvent = getOpenAIResponsesTerminalEvent(event); if (event.type === "response.created") { output.responseId = event.response.id; } else if (event.type === "response.output_item.added") { @@ -2297,8 +2311,8 @@ export async function processResponsesStream( closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id)); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } - } else if (event.type === "response.completed" || event.type === "response.incomplete") { - const response = event.response; + } else if (terminalEvent) { + const response = terminalEvent.response; finalizePendingResponsesToolCalls(output); if (response?.id) { output.responseId = response.id; @@ -2336,7 +2350,7 @@ export async function processResponsesStream( } promoteResponsesToolUseStopReason(output, (response as { end_turn?: boolean } | undefined)?.end_turn); options?.onCompleted?.(); - // `response.completed`/`response.incomplete` is the last event of a + // `response.completed`/`response.incomplete`/`response.done` is the last event of a // Responses stream. Stop pulling instead of waiting for the server to // close the connection: misbehaving providers keep the socket open // after the terminal event, which would park this loop until the idle diff --git a/packages/ai/test/openai-first-event-timeout.test.ts b/packages/ai/test/openai-first-event-timeout.test.ts index 34df11efe..4b5fcbbc1 100644 --- a/packages/ai/test/openai-first-event-timeout.test.ts +++ b/packages/ai/test/openai-first-event-timeout.test.ts @@ -705,6 +705,53 @@ describe("OpenAI-family first-event timeouts", () => { ]); }); + it("accepts response.done with a completed response as an OpenAI responses terminal event", async () => { + const completedResponse = createSseResponse([ + { type: "response.created", response: { id: "resp_done" } }, + { + type: "response.output_item.added", + item: { type: "message", id: "msg_done", role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", delta: "Hello done" }, + { + type: "response.output_item.done", + item: { + type: "message", + id: "msg_done", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "Hello done" }], + }, + }, + { + type: "response.done", + response: { + id: "resp_done", + status: "completed", + usage: { + input_tokens: 3, + output_tokens: 2, + total_tokens: 5, + }, + }, + }, + ]); + const fetch: FetchImpl = () => Promise.resolve(completedResponse); + const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), { + apiKey: "test-key", + fetch, + }).result(); + + expect(result.errorMessage).toBeUndefined(); + expect(result.stopReason).toBe("stop"); + expect(result.content as unknown[]).toContainEqual({ + type: "text", + text: "Hello done", + textSignature: '{"v":1,"id":"msg_done"}', + }); + }); + it("errors when Azure OpenAI responses stream closes without a terminal response event", async () => { const incompleteResponse = createSseResponse([ { type: "response.created", response: { id: "resp_incomplete_azure" } }, From 568226abb9b013870aa0d991dff24aa895f45004 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 08:49:02 +0000 Subject: [PATCH 50/91] fix(tui): disposed stale session ui renderers Stopped new-session and session-switch UI paths from detaching active loader/render components without running their disposal hooks. Added container and loader coverage for disposing children before destructive transcript/status replacement. Fixes #4686 --- .../modes/controllers/command-controller.ts | 36 ++++++---------- .../src/modes/controllers/event-controller.ts | 24 +++++------ .../controllers/extension-ui-controller.ts | 41 +++---------------- .../modes/controllers/selector-controller.ts | 5 +-- .../src/modes/interactive-mode.ts | 13 +++--- .../src/modes/utils/ui-helpers.ts | 4 +- packages/tui/src/tui.ts | 6 +++ packages/tui/test/container-dispose.test.ts | 11 +++++ packages/tui/test/loader.test.ts | 26 +++++++++++- 9 files changed, 82 insertions(+), 84 deletions(-) diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index 6fa6c1265..b6573e178 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -849,11 +849,7 @@ export class CommandController { } async #runNewSessionFlow(options?: NewSessionOptions, label: string = "New session started"): Promise { - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); + this.ctx.clearTransientSessionUi(); if (this.ctx.session.isCompacting) { this.ctx.session.abortCompaction(); @@ -867,14 +863,9 @@ export class CommandController { this.ctx.statusLine.invalidate(); this.ctx.statusLine.resetActiveTime(); - this.ctx.ui.requestRender(); this.ctx.updateEditorBorderColor(); - this.ctx.chatContainer.clear(); - this.ctx.pendingMessagesContainer.clear(); - this.ctx.compactionQueuedMessages = []; - this.ctx.streamingComponent = undefined; - this.ctx.streamingMessage = undefined; - this.ctx.pendingTools.clear(); + this.ctx.clearTransientSessionUi(); + this.ctx.resetTranscript(); this.ctx.present([new Spacer(1), new Text(`${theme.fg("accent", `${theme.status.success} ${label}`)}`, 1, 1)]); await this.ctx.reloadTodos(); @@ -914,7 +905,7 @@ export class CommandController { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; } - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const success = await this.ctx.session.fork(); if (!success) { @@ -1177,7 +1168,7 @@ export class CommandController { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; } - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const label = isAuto ? "Auto-compacting context... (esc to cancel)" : "Compacting context... (esc to cancel)"; const compactingLoader = new Loader( @@ -1207,7 +1198,7 @@ export class CommandController { await this.ctx.session.compact(instructions, options); compactingLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); this.ctx.rebuildChatFromMessages(); this.ctx.statusLine.invalidate(); @@ -1223,7 +1214,7 @@ export class CommandController { } } finally { compactingLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } // Run the caller's pre-flush hook (e.g. the plan-approval model transition) // before queued user input is dispatched, so any turn queued during @@ -1252,7 +1243,7 @@ export class CommandController { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; } - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const handoffLoader = new Loader( this.ctx.ui, @@ -1273,11 +1264,10 @@ export class CommandController { return; } - // Rebuild chat from the new session (which now contains the handoff document) - this.ctx.rebuildChatFromMessages(); - + // Rebuild chat from the new session (which now contains the handoff document). + this.ctx.clearTransientSessionUi(); + this.ctx.renderInitialMessages(); this.ctx.statusLine.invalidate(); - this.ctx.ui.requestRender(); this.ctx.updateEditorBorderColor(); await this.ctx.reloadTodos(); @@ -1297,9 +1287,9 @@ export class CommandController { } } finally { handoffLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } - this.ctx.ui.requestRender(); + this.ctx.ui.requestRender(true, { clearScrollback: true }); } } diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 88a66022b..e2baa993e 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -379,7 +379,7 @@ export class EventController { if (this.ctx.retryLoader) { this.ctx.retryLoader.stop(); this.ctx.retryLoader = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } this.#cancelIdleCompaction(); this.#cancelIdleRecap(); @@ -1083,7 +1083,7 @@ export class EventController { if (this.ctx.loadingAnimation) { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } if (this.ctx.streamingComponent) { this.ctx.chatContainer.removeChild(this.ctx.streamingComponent); @@ -1125,9 +1125,9 @@ export class EventController { /** * Tear down the live "Working…" loader: stop its animation timer AND clear the - * reference. A transient overlay (auto-compaction / auto-retry) that only ran - * `statusContainer.clear()` detached the loader from the container but left - * `ctx.loadingAnimation` set, so the resumed turn's `agent_start` → + * reference. A transient overlay (auto-compaction / auto-retry) can remove the + * loader from the container while leaving `ctx.loadingAnimation` set, so the + * resumed turn's `agent_start` → * `ensureLoadingAnimation()` (guarded by `if (!this.loadingAnimation)`) skipped * re-adding it and the spinner vanished while the agent kept streaming. Nulling * the reference here lets the next `agent_start` recreate and re-attach it. @@ -1168,7 +1168,7 @@ export class EventController { this.#cancelIdleRecap(); this.#setTerminalProgress(true); this.#stopWorkingLoader(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const reasonText = event.reason === "overflow" ? "Context overflow detected, " @@ -1203,7 +1203,7 @@ export class EventController { if (this.ctx.autoCompactionLoader) { this.ctx.autoCompactionLoader.stop(); this.ctx.autoCompactionLoader = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } const isHandoffAction = event.action === "handoff"; const isShakeAction = event.action === "shake"; @@ -1245,12 +1245,12 @@ export class EventController { } else if (event.errorMessage) { this.ctx.showWarning(event.errorMessage); } else if (isHandoffAction) { - this.ctx.chatContainer.clear(); + this.ctx.clearTransientSessionUi(); this.ctx.lastAssistantUsage = undefined; - this.ctx.rebuildChatFromMessages(); + this.ctx.renderInitialMessages(); this.ctx.statusLine.invalidate(); - this.ctx.ui.requestRender(); await this.ctx.reloadTodos(); + this.ctx.ui.requestRender(true, { clearScrollback: true }); this.ctx.showStatus("Auto-handoff completed"); } else if (event.skipped) { // Benign skip: no model selected, no candidate models available, or nothing @@ -1268,7 +1268,7 @@ export class EventController { async #handleAutoRetryStart(event: Extract): Promise { this.#trackRetrySupersededAssistantComponent(this.#lastAssistantComponent); this.#stopWorkingLoader(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); if (AIError.is(event.errorId, AIError.Flag.ThinkingLoop)) { // The retry path drops the failed assistant from runtime context. Do not // restore its inline Error row; just unpin the fixed-region banner so the @@ -1292,7 +1292,7 @@ export class EventController { if (this.ctx.retryLoader) { this.ctx.retryLoader.stop(); this.ctx.retryLoader = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } if (event.success) { let appliedRecovered = false; diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index 455f8de17..baea76ffd 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -162,18 +162,12 @@ export class ExtensionUiController { waitForIdle: () => this.ctx.session.agent.waitForIdle(), reload: async () => { await this.ctx.session.reload(); - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.showStatus("Reloaded session"); }, newSession: async options => { - // Stop any loading animation - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); + this.ctx.clearTransientSessionUi(); // Create new session this.clearExtensionTerminalInputListeners(); @@ -192,15 +186,8 @@ export class ExtensionUiController { // Reset and update status line this.ctx.statusLine.invalidate(); this.ctx.statusLine.resetActiveTime(); - this.ctx.ui.requestRender(); - - // Clear UI state - this.ctx.chatContainer.clear(); - this.ctx.pendingMessagesContainer.clear(); - this.ctx.compactionQueuedMessages = []; - this.ctx.streamingComponent = undefined; - this.ctx.streamingMessage = undefined; - this.ctx.pendingTools.clear(); + this.ctx.clearTransientSessionUi(); + this.ctx.resetTranscript(); this.ctx.present([ new Spacer(1), @@ -218,7 +205,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.editor.setText(result.selectedText); @@ -233,7 +219,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); if (result.editorText && !this.ctx.editor.getText().trim()) { @@ -251,7 +236,6 @@ export class ExtensionUiController { return { cancelled: true }; } setSessionTerminalTitle(this.ctx.sessionManager.getSessionName(), this.ctx.sessionManager.getCwd()); - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); return { cancelled: false }; @@ -398,18 +382,12 @@ export class ExtensionUiController { waitForIdle: () => this.ctx.session.agent.waitForIdle(), reload: async () => { await this.ctx.session.reload(); - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.showStatus("Reloaded session"); }, newSession: async options => { - // Stop any loading animation - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); + this.ctx.clearTransientSessionUi(); // Create new session this.clearExtensionTerminalInputListeners(); @@ -425,12 +403,8 @@ export class ExtensionUiController { } // Clear UI state - this.ctx.chatContainer.clear(); - this.ctx.pendingMessagesContainer.clear(); - this.ctx.compactionQueuedMessages = []; - this.ctx.streamingComponent = undefined; - this.ctx.streamingMessage = undefined; - this.ctx.pendingTools.clear(); + this.ctx.clearTransientSessionUi(); + this.ctx.resetTranscript(); this.ctx.present([ new Spacer(1), @@ -448,7 +422,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.editor.setText(result.selectedText); @@ -463,7 +436,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); if (result.editorText && !this.ctx.editor.getText().trim()) { @@ -480,7 +452,6 @@ export class ExtensionUiController { if (!result) { return { cancelled: true }; } - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); return { cancelled: false }; diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index 92b8a1ac7..d6e2a6d31 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -771,7 +771,6 @@ export class SelectorController { return; } - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); this.ctx.editor.setText(result.selectedText); done(); @@ -915,7 +914,6 @@ export class SelectorController { // Update UI — rebuild the display transcript for the new leaf (the // context from navigateTree is the LLM context, not the transcript). - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); if (result.editorText && !this.ctx.editor.getText().trim()) { @@ -927,7 +925,7 @@ export class SelectorController { } finally { if (summaryLoader) { summaryLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } this.ctx.editor.onEscape = originalOnEscape; } @@ -1067,7 +1065,6 @@ export class SelectorController { this.ctx.updateEditorBorderColor(); // Clear and re-render the chat - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.showStatus(movedProject ? `Resumed session in ${shortenPath(newCwd)}` : "Resumed session"); diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8e064d8ee..6be38c444 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -579,10 +579,10 @@ export class InteractiveMode implements InteractiveModeContext { this.retryLoader.stop(); this.retryLoader = undefined; } - this.statusContainer.clear(); - this.pendingMessagesContainer.clear(); + this.statusContainer.disposeChildren(); + this.pendingMessagesContainer.disposeChildren(); this.#cancelModelCycleClearTimer(); - this.modelCycleContainer.clear(); + this.modelCycleContainer.disposeChildren(); this.compactionQueuedMessages = []; this.streamingComponent = undefined; this.streamingMessage = undefined; @@ -3626,7 +3626,7 @@ export class InteractiveMode implements InteractiveModeContext { ensureLoadingAnimation(): void { if (!this.loadingAnimation) { this.#clearWorkingMessageAccentCache(); - this.statusContainer.clear(); + this.statusContainer.disposeChildren(); const messageColorFn = ((message: string) => renderWorkingMessage(message, this.#getWorkingMessageAccent())) as LoaderMessageColorFn & { animated?: true; @@ -3647,7 +3647,7 @@ export class InteractiveMode implements InteractiveModeContext { ); this.statusContainer.addChild(this.loadingAnimation); } else if (!this.statusContainer.children.includes(this.loadingAnimation)) { - this.statusContainer.clear(); + this.statusContainer.disposeChildren(); this.statusContainer.addChild(this.loadingAnimation); this.ui.requestRender(); } @@ -3660,7 +3660,7 @@ export class InteractiveMode implements InteractiveModeContext { this.loadingAnimation = undefined; this.#clearWorkingMessageAccentCache(); if (clearStatusContainer) { - this.statusContainer.clear(); + this.statusContainer.disposeChildren(); } } @@ -4123,7 +4123,6 @@ export class InteractiveMode implements InteractiveModeContext { } this.#btwController.dispose(); this.#omfgController.dispose(); - this.chatContainer.clear(); this.renderInitialMessages({ clearTerminalHistory: true }); this.updateEditorBorderColor(); this.showStatus( diff --git a/packages/coding-agent/src/modes/utils/ui-helpers.ts b/packages/coding-agent/src/modes/utils/ui-helpers.ts index d5f250824..bfc89ae00 100644 --- a/packages/coding-agent/src/modes/utils/ui-helpers.ts +++ b/packages/coding-agent/src/modes/utils/ui-helpers.ts @@ -581,7 +581,7 @@ export class UiHelpers { } else { this.ctx.resetTranscript(); } - this.ctx.pendingMessagesContainer.clear(); + this.ctx.pendingMessagesContainer.disposeChildren(); this.ctx.pendingBashComponents = []; this.ctx.pendingPythonComponents = []; @@ -647,7 +647,7 @@ export class UiHelpers { } updatePendingMessagesDisplay(): void { - this.ctx.pendingMessagesContainer.clear(); + this.ctx.pendingMessagesContainer.disposeChildren(); const queuedMessages = this.ctx.viewSession.getQueuedMessages() as QueuedMessages; const steeringMessages: Array<{ message: string; label: string }> = []; diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index bcf72f3b1..0d44beb4a 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -503,6 +503,12 @@ export class Container implements Component { this.#memoLines = undefined; } + /** Dispose every child, then detach it from this container. */ + disposeChildren(): void { + this.dispose(); + this.clear(); + } + invalidate(): void { this.#memoLines = undefined; for (const child of this.children) { diff --git a/packages/tui/test/container-dispose.test.ts b/packages/tui/test/container-dispose.test.ts index 7c497a2f6..a5d1da224 100644 --- a/packages/tui/test/container-dispose.test.ts +++ b/packages/tui/test/container-dispose.test.ts @@ -27,4 +27,15 @@ describe("Container.dispose", () => { outer.dispose(); expect(leafDisposed).toBe(1); }); + + it("disposeChildren disposes children and detaches them", () => { + let disposed = 0; + const container = new Container(); + container.addChild(inert(() => disposed++)); + + container.disposeChildren(); + + expect(disposed).toBe(1); + expect(container.children).toEqual([]); + }); }); diff --git a/packages/tui/test/loader.test.ts b/packages/tui/test/loader.test.ts index e200e1b24..e5e7967e8 100644 --- a/packages/tui/test/loader.test.ts +++ b/packages/tui/test/loader.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, setSystemTime, spyOn, vi } from "bun:test"; -import { TUI } from "@oh-my-pi/pi-tui"; +import { Container, TUI } from "@oh-my-pi/pi-tui"; import { Loader, type LoaderMessageColorFn } from "@oh-my-pi/pi-tui/components/loader"; import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; import { VirtualTerminal } from "./virtual-terminal"; @@ -144,4 +144,28 @@ describe("Loader component", () => { expect(() => loader.dispose()).not.toThrow(); // idempotent tui.stop(); }); + + it("container disposeChildren stops detached loader repaints", () => { + vi.useFakeTimers(); + const term = new VirtualTerminal(20, 4); + const tui = new TUI(term); + const spy = spyOn(tui, "requestComponentRender"); + const container = new Container(); + const loader = new Loader( + tui, + text => text, + text => text, + "Checking", + ["0", "1"], + ); + container.addChild(loader); + const afterMount = spy.mock.calls.length; + + container.disposeChildren(); + vi.advanceTimersByTime(200); + + expect(spy.mock.calls.length).toBe(afterMount); + expect(container.children).toEqual([]); + tui.stop(); + }); }); From 566069741e750d47855b83b590bc46acbc9d28e1 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 08:52:11 +0000 Subject: [PATCH 51/91] fix(auth): prioritized login api keys over env fallback - Marked API keys persisted by login so deliberate account switches outrank stale env fallbacks. - Kept ordinary stored API keys below explicit env vars and covered OpenCode Go regression. Fixes #4688 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/auth-storage.ts | 85 +++++++++++++------ .../test/auth-storage-api-key-login.test.ts | 26 ++++-- 3 files changed, 84 insertions(+), 31 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 0bf512a7a..89ff792ff 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenCode Go `/login` credentials being shadowed by an existing `OPENCODE_API_KEY` env fallback after switching accounts. ([#4688](https://github.com/can1357/oh-my-pi/issues/4688)) + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index a790ce5d5..d18ba42f3 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -66,6 +66,7 @@ const USAGE_RANKING_METRIC_EPSILON = 1e-9; export type ApiKeyCredential = { type: "api_key"; key: string; + source?: "login"; }; export type OAuthCredential = { @@ -1554,13 +1555,14 @@ export class AuthStorage { provider: string, type: T, sessionId?: string, + filter?: (credential: AuthCredential) => boolean, ): { credential: Extract; index: number } | undefined { const credentials = this.#getCredentialsForProvider(provider) .map((credential, index) => ({ credential, index })) - .filter( - (entry): entry is { credential: Extract; index: number } => - entry.credential.type === type, - ); + .filter((entry): entry is { credential: Extract; index: number } => { + if (entry.credential.type !== type) return false; + return filter?.(entry.credential) ?? true; + }); if (credentials.length === 0) return undefined; if (credentials.length === 1) return credentials[0]; @@ -1851,8 +1853,8 @@ export class AuthStorage { /** * Classify where a provider's auth comes from, following the same precedence * as {@link AuthStorage.getApiKey}: runtime override → config override → - * stored OAuth → env var → stored api_key → fallback resolver. Returns - * undefined when no auth is configured. + * stored OAuth → login-stored api_key → env var → stored api_key → + * fallback resolver. Returns undefined when no auth is configured. * * Compact, structured counterpart to {@link describeCredentialSource}. */ @@ -1861,6 +1863,9 @@ export class AuthStorage { if (this.#configOverrides.has(provider)) return { kind: "config" }; const stored = this.#getCredentialsForProvider(provider); if (stored.some(credential => credential.type === "oauth")) return { kind: "oauth" }; + if (stored.some(credential => credential.type === "api_key" && credential.source === "login")) { + return { kind: "api_key" }; + } if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) }; if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" }; if (this.#fallbackResolver?.(provider)) return { kind: "fallback" }; @@ -2005,7 +2010,7 @@ export class AuthStorage { if (!result) { return; } - const newCredential: ApiKeyCredential = { type: "api_key", key: result }; + const newCredential: ApiKeyCredential = { type: "api_key", key: result, source: "login" }; const stored = this.#store.upsertAuthCredentialRemote ? await this.#store.upsertAuthCredentialRemote(provider, newCredential) : this.#store.upsertAuthCredentialForProvider(provider, newCredential); @@ -3933,9 +3938,10 @@ export class AuthStorage { * 1. Runtime override (CLI --api-key) * 2. Config override (models.yml `providers..apiKey`) * 3. OAuth token from storage (auto-refreshed) - * 4. Environment variable - * 5. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins - * 6. Fallback resolver (models.yml custom providers, last-resort) + * 4. API key persisted by a successful `/login` + * 5. Environment variable + * 6. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins + * 7. Fallback resolver (models.yml custom providers, last-resort) */ async getApiKey(provider: string, sessionId?: string, options?: AuthApiKeyOptions): Promise { // Runtime override takes highest priority @@ -3954,13 +3960,24 @@ export class AuthStorage { return configKey; } - // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored - // static api_key (which may be a stale broker-migrated copy) as a last resort. + // Precedence: a deliberate OAuth/login credential wins, then an explicit env var, + // then a stored static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthResolved = await this.#resolveOAuthSelection(provider, sessionId, options); if (oauthResolved) { return oauthResolved.apiKey; } + const loginApiKeySelection = this.#selectCredentialByType( + provider, + "api_key", + sessionId, + credential => credential.type === "api_key" && credential.source === "login", + ); + if (loginApiKeySelection) { + this.#recordSessionCredential(provider, sessionId, "api_key", loginApiKeySelection.index); + return this.#configValueResolver(loginApiKeySelection.credential.key); + } + // Past OAuth: the session sticky (if any) is stale — the request authenticates via // env/api_key/fallback, not OAuth, so clear it now so getOAuthAccountId() correctly // suppresses account_uuid for this session. @@ -3969,7 +3986,12 @@ export class AuthStorage { const envKey = getEnvApiKey(provider); if (envKey) return envKey; - const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId); + const apiKeySelection = this.#selectCredentialByType( + provider, + "api_key", + sessionId, + credential => credential.type !== "api_key" || credential.source !== "login", + ); if (apiKeySelection) { this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index); return this.#configValueResolver(apiKeySelection.credential.key); @@ -4674,9 +4696,10 @@ export class AuthStorage { * 1. Runtime override (`--api-key`). * 2. Config override (`models.yml` `providers..apiKey`). * 3. Stored OAuth credential. - * 4. Env var — overrides a stored static api_key (e.g. a stale broker copy). - * 5. Stored api_key credential. - * 6. Fallback resolver. + * 4. API key persisted by a successful `/login`. + * 5. Env var — overrides a stored static api_key (e.g. a stale broker copy). + * 6. Stored api_key credential. + * 7. Fallback resolver. * * The string is purely informational; consumers must not parse it. */ @@ -4691,14 +4714,16 @@ export class AuthStorage { const baseLabel = this.#sourceLabel ?? "local store"; const stored = this.#getStoredCredentials(provider); const session = sessionId ? this.#sessionLastCredential.get(provider)?.get(sessionId) : undefined; - // Describe the stored credential of a given type, honoring the session sticky index. - const describeStored = (type: AuthCredential["type"]): string | undefined => { + const describeStored = ( + type: AuthCredential["type"], + filter?: (credential: AuthCredential) => boolean, + ): string | undefined => { const typed = stored .map((entry, index) => ({ entry, index })) - .filter(({ entry }) => entry.credential.type === type); + .filter(({ entry }) => entry.credential.type === type && (filter?.(entry.credential) ?? true)); if (typed.length === 0) return undefined; - const index = session?.type === type ? session.index : typed[0].index; - const chosen = stored[index] ?? typed[0].entry; + const sticky = session?.type === type ? typed.find(entry => entry.index === session.index) : undefined; + const chosen = sticky?.entry ?? typed[0].entry; const credential = chosen.credential; const identity = credential.type === "oauth" @@ -4707,11 +4732,19 @@ export class AuthStorage { return `${baseLabel} · ${type} #${chosen.id} (${identity})`; }; - // A deliberate OAuth login wins; then an explicit env var; then a stored static api_key. + // Deliberate login credentials win; then an explicit env var; then a stored static api_key. const oauthSource = describeStored("oauth"); if (oauthSource) return oauthSource; + const loginApiKeySource = describeStored( + "api_key", + credential => credential.type === "api_key" && credential.source === "login", + ); + if (loginApiKeySource) return loginApiKeySource; if (getEnvApiKey(provider)) return `env (over ${baseLabel})`; - const apiKeySource = describeStored("api_key"); + const apiKeySource = describeStored( + "api_key", + credential => credential.type !== "api_key" || credential.source !== "login", + ); if (apiKeySource) return apiKeySource; if (this.#fallbackResolver?.(provider) !== undefined) return "fallback resolver"; return undefined; @@ -4776,9 +4809,10 @@ function normalizeStoredIdentityKey(identityKey: string | null | undefined): str function serializeCredential(provider: string, credential: AuthCredential): SerializedCredentialRecord | null { if (credential.type === "api_key") { + const data = credential.source === "login" ? { key: credential.key, source: "login" } : { key: credential.key }; return { credentialType: "api_key", - data: JSON.stringify({ key: credential.key }), + data: JSON.stringify(data), identityKey: null, }; } @@ -4806,7 +4840,8 @@ function deserializeCredential(row: AuthRow): AuthCredential | null { if (row.credential_type === "api_key") { const data = parsed as Record; if (typeof data.key === "string") { - return { type: "api_key", key: data.key }; + const source = data.source === "login" ? "login" : undefined; + return source ? { type: "api_key", key: data.key, source } : { type: "api_key", key: data.key }; } } if (row.credential_type === "oauth") { diff --git a/packages/ai/test/auth-storage-api-key-login.test.ts b/packages/ai/test/auth-storage-api-key-login.test.ts index 879a8a9ec..5d4fe3c39 100644 --- a/packages/ai/test/auth-storage-api-key-login.test.ts +++ b/packages/ai/test/auth-storage-api-key-login.test.ts @@ -39,9 +39,9 @@ function countCredentialRowsByDisabledState(dbPath: string, provider: string, di } describe("AuthStorage api-key login upsert", () => { - // A live env var now (correctly) overrides a stored static api_key. These tests verify that a - // freshly stored api_key resolves through AuthStorage.getApiKey, so neutralize the env leg - // entirely — this ignores every provider's ambient env key, not just the few set locally. + // Most tests neutralize the env leg so ambient shell / ~/.env keys cannot + // hide the stored credential behavior under test. Login-persisted API keys + // have their own precedence coverage below. let tempDir = ""; let dbPath = ""; let store: SqliteAuthCredentialStore | null = null; @@ -49,9 +49,10 @@ describe("AuthStorage api-key login upsert", () => { let loginDeepSeekSpy: Mock; let loginKagiSpy: Mock; let loginOllamaCloudSpy: Mock; + let getEnvApiKeySpy: Mock; beforeEach(async () => { - vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); + getEnvApiKeySpy = vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-api-key-login-")); dbPath = path.join(tempDir, "agent.db"); store = await SqliteAuthCredentialStore.open(dbPath); @@ -118,8 +119,8 @@ describe("AuthStorage api-key login upsert", () => { const credentials = store.listAuthCredentials("kagi"); expect(credentials.map(entry => entry.credential)).toEqual([ - { type: "api_key", key: "first-kagi-key" }, - { type: "api_key", key: "second-kagi-key" }, + { type: "api_key", key: "first-kagi-key", source: "login" }, + { type: "api_key", key: "second-kagi-key", source: "login" }, ]); const rotatedKeys = [await authStorage.getApiKey("kagi"), await authStorage.getApiKey("kagi")].sort(); expect(rotatedKeys).toEqual(["first-kagi-key", "second-kagi-key"]); @@ -188,4 +189,17 @@ describe("AuthStorage api-key login upsert", () => { expect(store.getApiKey("deepseek")).toBe("same-deepseek-key"); expect(await authStorage.getApiKey("deepseek", "session-deepseek-relogin")).toBe("same-deepseek-key"); }); + + it("uses a fresh OpenCode Go login over an existing env fallback", async () => { + if (!authStorage) throw new Error("test setup failed"); + + getEnvApiKeySpy.mockImplementation(provider => (provider === "opencode-go" ? "old-opencode-key" : undefined)); + + await authStorage.login("opencode-go", { + onAuth: () => {}, + onPrompt: async () => "new-opencode-key", + }); + + expect(await authStorage.getApiKey("opencode-go", "session-opencode-go-login")).toBe("new-opencode-key"); + }); }); From bfb170ae692bc1e0fa7867d2a0e1ae6aabcb3825 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 09:41:24 +0000 Subject: [PATCH 52/91] fix(catalog): fixed litellm bundled catalog fallback Resolved LiteLLM dynamic discovery to fall back to bundled catalog references when models.dev has no matching model. Added rich-endpoint and /v1/models fallback regressions for glm-5.2 reasoning/thinking metadata. Fixes #4695 --- packages/catalog/CHANGELOG.md | 4 + .../src/provider-models/openai-compat.ts | 6 +- .../catalog/test/litellm-provider.test.ts | 97 +++++++++++++++++++ 3 files changed, 104 insertions(+), 3 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index e4913294f..192978bd1 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed LiteLLM discovery to fall back to bundled catalog metadata when `models.dev` lacks a model reference, preserving reasoning and thinking support for models such as `glm-5.2`. ([#4695](https://github.com/can1357/oh-my-pi/issues/4695)) + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index d43a20143..6be4c1f24 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3346,11 +3346,11 @@ export function litellmModelManagerOptions( cacheProviderId: `litellm:rich-v3:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy and is never bundled in models.json (that // would leak the machine's localhost catalog). Prefer the proxy's richer - // management metadata, then fall back to /v1/models and enrich bare ids - // against models.dev like the gateway providers (fireworks et al.) do. + // management metadata, then enrich ids against models.dev with the bundled + // catalog as a fallback before using /v1/models. fetchDynamicModels: async () => { const modelsDevReferences = await loadModelsDevReferences<"openai-completions">(config?.fetch); - const resolveReference = (id: string) => modelsDevReferences.get(id); + const resolveReference = createReferenceResolver(modelsDevReferences); const richModels = await fetchLiteLLMRichModels({ api: "openai-completions", provider: "litellm", diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index ec6d88704..dd9e042c6 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -238,6 +238,54 @@ describe("LiteLLM provider discovery", () => { }); }); + test("enriches LiteLLM rich models missing from models.dev with bundled reasoning metadata", async () => { + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ + data: [ + { + model_group: "glm-5.2", + model_name: "GLM-5.2", + }, + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when model_group info has a real model"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(models).toHaveLength(1); + expect(models?.[0]).toMatchObject({ + id: "glm-5.2", + name: "GLM-5.2", + api: "openai-completions", + provider: "litellm", + baseUrl: "http://primary:4000/v1", + reasoning: true, + thinking: { + mode: "effort", + efforts: ["minimal", "low", "medium", "high", "xhigh"], + effortMap: { + minimal: "none", + xhigh: "max", + }, + }, + }); + }); + test("uses LiteLLM tool support metadata when rich endpoints succeed", async () => { const fetchMock = vi.fn(async (input: string | URL | Request) => { const url = inputUrl(input); @@ -547,4 +595,53 @@ describe("LiteLLM provider discovery", () => { maxTokens: 8_192, }); }); + + test("enriches LiteLLM /v1/models fallback entries missing from models.dev with bundled reasoning metadata", async () => { + const calls: string[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + calls.push(url); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if ( + url === "http://primary:4000/model_group/info" || + url === "http://primary:4000/v2/model/info" || + url === "http://primary:4000/model/info" || + url === "http://primary:4000/v1/model/info" + ) { + return new Response("{}", { status: 404 }); + } + if (url === "http://primary:4000/v1/models") { + return Response.json({ data: [{ id: "glm-5.2" }] }); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-fallback", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(calls).toContain("http://primary:4000/v1/models"); + expect(models).toHaveLength(1); + expect(models?.[0]).toMatchObject({ + id: "glm-5.2", + name: "GLM-5.2", + api: "openai-completions", + provider: "litellm", + baseUrl: "http://primary:4000/v1", + reasoning: true, + thinking: { + mode: "effort", + efforts: ["minimal", "low", "medium", "high", "xhigh"], + effortMap: { + minimal: "none", + xhigh: "max", + }, + }, + }); + }); }); From f44fcdfb7df103b2053b226664aaed82fddf2961 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 11:04:51 +0000 Subject: [PATCH 53/91] fix(coding-agent): hid marketplace clone temp paths Displayed cloned marketplace catalog validation errors with repository-relative catalog paths and the original source identifier. Fixes #4702 --- .../plugins/marketplace/fetcher.ts | 29 ++++++++++--------- .../test/marketplace/fetcher.test.ts | 21 +++++++++++++- 2 files changed, 35 insertions(+), 15 deletions(-) diff --git a/packages/coding-agent/src/extensibility/plugins/marketplace/fetcher.ts b/packages/coding-agent/src/extensibility/plugins/marketplace/fetcher.ts index 28d342fce..6a6107343 100644 --- a/packages/coding-agent/src/extensibility/plugins/marketplace/fetcher.ts +++ b/packages/coding-agent/src/extensibility/plugins/marketplace/fetcher.ts @@ -196,19 +196,20 @@ export function parseMarketplaceCatalog(content: string, filePath: string): Mark * Catalog paths tried in priority order: omp-namespaced override first, then * the Claude Code-compatible fallback so existing marketplaces keep loading. */ -const CATALOG_RELATIVE_PATHS: readonly string[] = [ - path.join(".omp-plugin", "marketplace.json"), - path.join(".claude-plugin", "marketplace.json"), -]; +const CATALOG_RELATIVE_PATHS: readonly string[] = [".omp-plugin/marketplace.json", ".claude-plugin/marketplace.json"]; -async function readMarketplaceCatalog(root: string): Promise<{ catalogPath: string; content: string }> { +async function readMarketplaceCatalog( + root: string, + options: { relativeDisplayPaths?: boolean } = {}, +): Promise<{ catalogPath: string; displayPath: string; content: string }> { const tried: string[] = []; for (const rel of CATALOG_RELATIVE_PATHS) { - const catalogPath = path.join(root, rel); - tried.push(catalogPath); + const catalogPath = path.join(root, ...rel.split("/")); + const displayPath = options.relativeDisplayPaths ? rel : catalogPath; + tried.push(displayPath); try { const content = await Bun.file(catalogPath).text(); - return { catalogPath, content }; + return { catalogPath, displayPath, content }; } catch (err) { if (isEnoent(err)) continue; throw err; @@ -252,11 +253,11 @@ export async function fetchMarketplace(source: string, cacheDir: string): Promis if (type === "github") { const url = `https://github.com/${source}.git`; - return cloneAndReadCatalog(url, cacheDir); + return cloneAndReadCatalog(url, source, cacheDir); } if (type === "git") { - return cloneAndReadCatalog(source, cacheDir); + return cloneAndReadCatalog(source, source, cacheDir); } // type === "url" @@ -284,7 +285,7 @@ export async function fetchMarketplace(source: string, cacheDir: string): Promis * responsible for promoting the clone to its final cache location via * `promoteCloneToCache` after any duplicate/drift checks pass. */ -async function cloneAndReadCatalog(url: string, cacheDir: string): Promise { +async function cloneAndReadCatalog(url: string, source: string, cacheDir: string): Promise { const tmpDir = path.join(cacheDir, `.tmp-clone-${Date.now()}`); await fs.mkdir(cacheDir, { recursive: true }); @@ -292,12 +293,12 @@ async function cloneAndReadCatalog(url: string, cacheDir: string): Promise {}); - throw new Error(`Cloned repository ${url}: ${(err as Error).message}`, { cause: err }); + throw new Error(`Cloned repository ${url}: ${(err as Error).message} (source: ${source})`, { cause: err }); } } diff --git a/packages/coding-agent/test/marketplace/fetcher.test.ts b/packages/coding-agent/test/marketplace/fetcher.test.ts index e72c84f1e..5e73e0517 100644 --- a/packages/coding-agent/test/marketplace/fetcher.test.ts +++ b/packages/coding-agent/test/marketplace/fetcher.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, spyOn } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -7,6 +7,7 @@ import { fetchMarketplace, parseMarketplaceCatalog, } from "@oh-my-pi/pi-coding-agent/extensibility/plugins/marketplace"; +import * as git from "@oh-my-pi/pi-coding-agent/utils/git"; import { removeSyncWithRetries } from "@oh-my-pi/pi-utils"; // Fixture lives at test/marketplace/fixtures/valid-marketplace/ @@ -231,6 +232,24 @@ describe("fetchMarketplace", () => { ); }); + it("hides temp clone paths in cloned catalog validation errors", async () => { + const cloneSpy = spyOn(git, "clone").mockImplementation(async (_url, targetDir) => { + fs.mkdirSync(path.join(targetDir, ".claude-plugin"), { recursive: true }); + fs.writeFileSync( + path.join(targetDir, ".claude-plugin", "marketplace.json"), + JSON.stringify({ name: "broken-marketplace", plugins: [] }), + ); + }); + + try { + await expect(fetchMarketplace("kubeshark/kubeshark", tmpDir)).rejects.toThrow( + 'Cloned repository https://github.com/kubeshark/kubeshark.git: Missing or invalid field "owner" in catalog: .claude-plugin/marketplace.json (source: kubeshark/kubeshark)', + ); + } finally { + cloneSpy.mockRestore(); + } + }); + // Network-dependent tests — skip in CI / offline environments. // These verify real git clone and HTTP fetch error handling. it.skip("github source throws on nonexistent repo", async () => { From 389515baadae9f3ea2d3400b2c18b22bf76fb09c Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 14:13:02 +0000 Subject: [PATCH 54/91] fix(agent): retried handoff auto-only tool choice errors Retried handoff generation with toolChoice auto when a provider rejects the cache-preserving toolChoice none request as auto-only. Kept unrelated provider 400s terminal so bad request failures still surface without masking the cause. Fixes #4715 --- packages/agent/CHANGELOG.md | 4 + packages/agent/src/compaction/compaction.ts | 39 +++++-- packages/agent/test/handoff.test.ts | 119 +++++++++++++++++++- 3 files changed, 149 insertions(+), 13 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 1de3328af..89442e470 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed handoff generation retrying with `toolChoice: "auto"` when custom OpenAI-compatible providers reject `toolChoice: "none"` with an auto-only 400. ([#4715](https://github.com/can1357/oh-my-pi/issues/4715)) + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 9f502bac5..6459dc829 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -698,6 +698,12 @@ function createSummarizationError(prefix: string, response: AssistantMessage): E return response.errorStatus === undefined ? new Error(text) : new ProviderHttpError(text, response.errorStatus); } +function shouldRetryHandoffWithAutoToolChoice(response: AssistantMessage): boolean { + if (response.errorStatus !== 400) return false; + const message = response.errorMessage ?? ""; + return /\btool_choice\b/i.test(message) && /\bauto\b/i.test(message) && /\bsupported\b/i.test(message); +} + /** * Generate a summary of the conversation using the LLM. * If previousSummary is provided, uses the update prompt to merge. @@ -917,24 +923,33 @@ export interface HandoffFromContextOptions { * `streamOptions` that mirror the live turn's cache routing. That keeps the * cache-preserving context construction in the host (which owns the transform * pipeline) while this function centralizes the handoff request contract: - * `toolChoice: "none"`, clamped reasoning effort, oneshot telemetry, text-only - * extraction, and provider-error mapping. + * cache-first `toolChoice: "none"`, clamped reasoning effort, one retry for + * auto-only `tool_choice` providers, oneshot telemetry, text-only extraction, + * and provider-error mapping. */ export async function generateHandoffFromContext( context: Context, model: Model, options: HandoffFromContextOptions, ): Promise { - const response = await instrumentedCompleteSimple( - model, - context, - { - ...options.streamOptions, - reasoning: resolveCompactionEffort(model, options.thinkingLevel), - toolChoice: "none", - }, - { telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl }, - ); + const requestOptions = { + ...options.streamOptions, + reasoning: resolveCompactionEffort(model, options.thinkingLevel), + toolChoice: "none" as const, + }; + let response = await instrumentedCompleteSimple(model, context, requestOptions, { + telemetry: options.telemetry, + oneshotKind: "handoff", + completeImpl: options.completeImpl, + }); + if (response.stopReason === "error" && shouldRetryHandoffWithAutoToolChoice(response)) { + response = await instrumentedCompleteSimple( + model, + context, + { ...requestOptions, toolChoice: "auto" }, + { telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl }, + ); + } if (response.stopReason === "error") { throw createSummarizationError("Handoff generation failed", response); diff --git a/packages/agent/test/handoff.test.ts b/packages/agent/test/handoff.test.ts index a3bdc293c..cf89f4344 100644 --- a/packages/agent/test/handoff.test.ts +++ b/packages/agent/test/handoff.test.ts @@ -9,7 +9,7 @@ import { import { ThinkingLevel } from "@oh-my-pi/pi-agent-core/thinking"; import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai"; import * as ai from "@oh-my-pi/pi-ai"; -import { Effort } from "@oh-my-pi/pi-ai"; +import { Effort, z } from "@oh-my-pi/pi-ai"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; function createAssistantMessage(content: AssistantMessage["content"]): AssistantMessage { @@ -32,6 +32,28 @@ function createAssistantMessage(content: AssistantMessage["content"]): Assistant }; } +function createAssistantError(errorStatus: number, errorMessage: string): AssistantMessage { + return { + ...createAssistantMessage([]), + stopReason: "error", + errorStatus, + errorMessage, + }; +} + +const handoffToolSchema = z.object({ note: z.string().optional() }); + +function createHandoffTool(): AgentTool { + return { + name: "handoff_probe", + label: "Handoff Probe", + description: "Confirms handoff requests keep live tools available.", + parameters: handoffToolSchema, + intent: "omit", + execute: async () => ({ content: [{ type: "text", text: "ok" }], details: {} }), + }; +} + function getTestModel(): Model { const model = getBundledModel("anthropic", "claude-sonnet-4-5"); if (!model) { @@ -155,4 +177,99 @@ describe("handoff helpers", () => { reasoning: Effort.Medium, }); }); + + test("generateHandoffFromContext retries auto-only tool_choice rejection with live tools", async () => { + const completeSimpleSpy = vi + .spyOn(ai, "completeSimple") + .mockResolvedValueOnce( + createAssistantError( + 400, + "400 Bad Request: Only a tool_choice of 'auto' is supported for this model; param=tool_choice", + ), + ) + .mockResolvedValueOnce(createAssistantMessage([{ type: "text", text: "## Goal\nRecovered on retry" }])); + const model = getTestModel(); + const tools = [createHandoffTool()]; + const context = { + systemPrompt: ["Live system prompt"], + tools, + messages: [{ role: "user" as const, content: "prepare handoff", timestamp: 1 }], + }; + + const document = await generateHandoffFromContext(context, model, { + streamOptions: { + apiKey: "test-key", + sessionId: "sess-auto-only:side:42", + promptCacheKey: "sess-auto-only", + }, + thinkingLevel: ThinkingLevel.Medium, + }); + + expect(document).toBe("## Goal\nRecovered on retry"); + expect(completeSimpleSpy).toHaveBeenCalledTimes(2); + const firstCall = completeSimpleSpy.mock.calls[0]; + const secondCall = completeSimpleSpy.mock.calls[1]; + if (!firstCall) throw new Error("Expected initial completeSimple call"); + if (!secondCall) throw new Error("Expected retry completeSimple call"); + const [firstModel, firstContext, firstOptions] = firstCall; + const [secondModel, secondContext, secondOptions] = secondCall; + expect(firstModel).toBe(model); + expect(secondModel).toBe(model); + expect(firstContext).toBe(context); + expect(secondContext).toBe(context); + expect(firstContext.tools).toBe(tools); + expect(secondContext.tools).toBe(tools); + expect(firstOptions).toMatchObject({ + apiKey: "test-key", + sessionId: "sess-auto-only:side:42", + promptCacheKey: "sess-auto-only", + toolChoice: "none", + reasoning: Effort.Medium, + }); + expect(secondOptions).toMatchObject({ + apiKey: "test-key", + sessionId: "sess-auto-only:side:42", + promptCacheKey: "sess-auto-only", + toolChoice: "auto", + reasoning: Effort.Medium, + }); + }); + + test("generateHandoffFromContext surfaces unrelated provider 400 without retrying", async () => { + const completeSimpleSpy = vi + .spyOn(ai, "completeSimple") + .mockResolvedValueOnce(createAssistantError(400, "400 Bad Request: unsupported max_tokens; param=max_tokens")); + const model = getTestModel(); + const tools = [createHandoffTool()]; + const context = { + systemPrompt: ["Live system prompt"], + tools, + messages: [{ role: "user" as const, content: "prepare handoff", timestamp: 1 }], + }; + + const error = await generateHandoffFromContext(context, model, { + streamOptions: { + apiKey: "test-key", + sessionId: "sess-unrelated-400:side:42", + promptCacheKey: "sess-unrelated-400", + }, + thinkingLevel: ThinkingLevel.Medium, + }).catch((caught: unknown) => caught); + + if (!(error instanceof Error)) throw new Error("Expected handoff generation to reject"); + expect(error.message).toContain("unsupported max_tokens"); + expect(completeSimpleSpy).toHaveBeenCalledTimes(1); + const call = completeSimpleSpy.mock.calls[0]; + if (!call) throw new Error("Expected completeSimple call"); + const [, calledContext, options] = call; + expect(calledContext).toBe(context); + expect(calledContext.tools).toBe(tools); + expect(options).toMatchObject({ + apiKey: "test-key", + sessionId: "sess-unrelated-400:side:42", + promptCacheKey: "sess-unrelated-400", + toolChoice: "none", + reasoning: Effort.Medium, + }); + }); }); From 4b29248877585dc4ee1948a3608daad184127f6e Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 15:23:59 +0000 Subject: [PATCH 55/91] fix(cli): stopped exporting launch env local values Filtered Bun-autoloaded launch .env.local entries out of child shell environments so nested commands can load their own dotenv files. Added a regression test covering Convex-style inherited deployment variables while preserving ordinary inherited env values. Fixes #4723 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../test/non-interactive-env.test.ts | 54 +++++++++++++++++++ packages/utils/CHANGELOG.md | 4 ++ packages/utils/src/env.ts | 13 +++++ packages/utils/src/procmgr.ts | 4 +- 5 files changed, 77 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ec2a2142b..8b05781d0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed bash/tool command environments inheriting Bun-autoloaded launch `.env.local` values, so nested apps can load their own dotenv values without parent deployment variables taking precedence. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) + ## [16.3.10] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/test/non-interactive-env.test.ts b/packages/coding-agent/test/non-interactive-env.test.ts index 57c4ca878..669a0b4db 100644 --- a/packages/coding-agent/test/non-interactive-env.test.ts +++ b/packages/coding-agent/test/non-interactive-env.test.ts @@ -1,4 +1,7 @@ import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { buildNonInteractiveEnv } from "@oh-my-pi/pi-coding-agent/exec/non-interactive-env"; describe("buildNonInteractiveEnv", () => { @@ -45,3 +48,54 @@ describe("buildNonInteractiveEnv", () => { expect(env.LC_ALL).toBeUndefined(); }); }); + +it("keeps launch .env.local values out of child shell config", async () => { + const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "omp-env-local-")); + try { + await Bun.write( + path.join(tmp, ".env.local"), + "CONVEX_DEPLOYMENT=anonymous:root-local\nCONVEX_URL=http://127.0.0.1:3210\n", + ); + const procmgrPath = path.resolve(import.meta.dir, "../../utils/src/procmgr.ts"); + const script = [ + `import { getShellConfig } from ${JSON.stringify(procmgrPath)};`, + "const env = getShellConfig().env;", + "console.log(JSON.stringify({", + " deployment: env.CONVEX_DEPLOYMENT ?? null,", + " url: env.CONVEX_URL ?? null,", + " inherited: env.OMP_TEST_INHERITED_MARKER ?? null,", + "}));", + ].join("\n"); + const proc = Bun.spawn([process.execPath, "--no-install", "--eval", script], { + cwd: tmp, + env: { + HOME: process.env.HOME ?? "", + OMP_TEST_INHERITED_MARKER: "keep-me", + PATH: process.env.PATH ?? "", + SHELL: process.env.SHELL ?? "/bin/bash", + }, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + proc.exited, + ]); + + expect(stderr).toBe(""); + expect(exitCode).toBe(0); + const payload: { + deployment: string | null; + url: string | null; + inherited: string | null; + } = JSON.parse(stdout); + expect(payload).toEqual({ + deployment: null, + url: null, + inherited: "keep-me", + }); + } finally { + await fs.rm(tmp, { recursive: true, force: true }); + } +}); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 80ebafb3a..aa19564d2 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed child shell environment filtering to drop launch-directory `.env.local` values that Bun auto-loaded before OMP starts command shells. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) + ## [16.3.10] - 2026-07-06 ### Added diff --git a/packages/utils/src/env.ts b/packages/utils/src/env.ts index b3337f92e..21619ae2d 100644 --- a/packages/utils/src/env.ts +++ b/packages/utils/src/env.ts @@ -53,6 +53,19 @@ export function filterProcessEnv(env: Record): Recor return result; } +/** Filters process env for child shells without launch-cwd `.env.local` values. */ +export function filterChildShellEnv( + env: Record, + cwd: string = process.cwd(), +): Record { + const result = filterProcessEnv(env); + const launchLocalEnv = parseEnvFile(path.join(cwd, ".env.local")); + for (const key in launchLocalEnv) { + if (result[key] === launchLocalEnv[key]) delete result[key]; + } + return result; +} + /** * Parses a .env file synchronously and extracts key-value string pairs. * Ignores lines that are empty or start with '#'. Trims whitespace. diff --git a/packages/utils/src/procmgr.ts b/packages/utils/src/procmgr.ts index 3097900b1..5a4cb30b9 100644 --- a/packages/utils/src/procmgr.ts +++ b/packages/utils/src/procmgr.ts @@ -2,7 +2,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { Process, ProcessStatus } from "@oh-my-pi/pi-natives"; import type { Subprocess } from "bun"; -import { $env, filterProcessEnv } from "./env"; +import { $env, filterChildShellEnv } from "./env"; import { $which } from "./which"; export interface ShellConfig { @@ -31,7 +31,7 @@ export function isExecutable(path: string): boolean { function buildSpawnEnv(shell: string): Record { const noCI = $env.PI_BASH_NO_CI || $env.CLAUDE_BASH_NO_CI; return { - ...filterProcessEnv(Bun.env), + ...filterChildShellEnv(Bun.env), SHELL: shell, GIT_EDITOR: "true", GPG_TTY: "not a tty", From a749d46ef67f3c10934425e9607fab57f0f80016 Mon Sep 17 00:00:00 2001 From: Jeff Scott Ward Date: Mon, 6 Jul 2026 11:24:52 -0400 Subject: [PATCH 56/91] fix: preserve status line path under overflow --- .../modes/components/status-line/component.ts | 16 ++- .../test/status-line-overflow.test.ts | 103 ++++++++++++++++-- 2 files changed, 108 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts index 380e0d5e6..195539a46 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.ts @@ -1235,9 +1235,21 @@ export class StatusLineComponent implements Component { } } } + const leftOverflowDropIndex = (): number => { + // Preserve the current working directory as long as possible. The + // previous right-to-left pop could collapse a normal-width bar to + // just the model segment, hiding the path before less-critical left + // segments such as model/mode/collab were removed. + for (let i = leftSegIds.length - 1; i >= 0; i--) { + if (leftSegIds[i] !== "path") return i; + } + return left.length - 1; + }; + while (totalWidth() > topFillWidth && left.length > 0) { - left.pop(); - leftSegIds.pop(); + const dropIdx = leftOverflowDropIndex(); + left.splice(dropIdx, 1); + leftSegIds.splice(dropIdx, 1); leftWidth = groupWidth(left, leftCapWidth, leftSepWidth); } } diff --git a/packages/coding-agent/test/status-line-overflow.test.ts b/packages/coding-agent/test/status-line-overflow.test.ts index fd6307971..b73b9d230 100644 --- a/packages/coding-agent/test/status-line-overflow.test.ts +++ b/packages/coding-agent/test/status-line-overflow.test.ts @@ -77,13 +77,27 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null }; } -function createStatusLineSession(sessionName: string) { +function createStatusLineSession(sessionName: string, modelName?: string) { + const model = modelName ? { name: modelName, contextWindow: 128000 } : undefined; return { - state: { messages: [] }, + state: { messages: [], model }, + messages: [], + model: model ?? { contextWindow: 128000 }, + contextUsageRevision: 0, + systemPrompt: [], + agent: { state: { tools: [] } }, + skills: [], isStreaming: false, + isAutoThinking: false, + autoResolvedThinkingLevel: () => undefined, + isAdvisorActive: () => false, + isFastModeActive: () => false, getAsyncJobSnapshot: () => ({ running: [] }), getCurrentModel: () => undefined, isFastModeEnabled: () => false, + getContextUsage: () => ({ tokens: 0, contextWindow: 128000 }), + getGoalModeState: () => null, + modelRegistry: { isUsingOAuth: () => false }, sessionManager: { getSessionName: () => sessionName, getUsageStatistics: () => ({ @@ -102,6 +116,10 @@ function createStatusLineSession(sessionName: string) { } as unknown as ConstructorParameters[0]; } +function stripAnsi(value: string): string { + return value.replace(/\x1B\[[0-?]*[ -/]*[@-~]/g, ""); +} + describe("status line session accent", () => { function buildComponent(sessionAccent: boolean) { const component = new StatusLineComponent(createStatusLineSession("Named session")); @@ -244,10 +262,17 @@ describe("overflow: path shrinks before git is dropped", () => { } } - // Left-pop loop (fallback) + // Left-segment fallback loop. + const leftOverflowDropIndex = (): number => { + for (let i = leftSegIds.length - 1; i >= 0; i--) { + if (leftSegIds[i] !== "path") return i; + } + return left.length - 1; + }; while (groupWidth() > width && left.length > 0) { - left.pop(); - leftSegIds.pop(); + const dropIdx = leftOverflowDropIndex(); + left.splice(dropIdx, 1); + leftSegIds.splice(dropIdx, 1); } return { surviving: [...leftSegIds], contents: [...left] }; @@ -285,18 +310,20 @@ describe("overflow: path shrinks before git is dropped", () => { }); it("shrinks a short path when maxLength exceeds actual path length", () => { - // Short dir name — rendered path is well under maxLength=80 + // Short dir name — rendered path is well under the configured maxLength. const shortDir = fs.mkdtempSync(path.join(os.tmpdir(), "omp-short-")); setProjectDir(shortDir); try { - const ctx = createCtx({ pathMaxLength: 80, branch: "feat/long-branch-name" }); + const maxLength = 160; + const ctx = createCtx({ pathMaxLength: maxLength, branch: "feat/long-branch-name" }); const fullPath = renderSegment("path", ctx); const fullGit = renderSegment("git", ctx); const pathVW = visibleWidth(fullPath.content); const gitVW = visibleWidth(fullGit.content); - // Sanity: path is shorter than maxLength — this is the bug scenario - expect(pathVW).toBeLessThan(80); + // Sanity: path is shorter than maxLength — this is the bug scenario. + // macOS temp paths can exceed 80 columns once the path icon is included. + expect(pathVW).toBeLessThan(maxLength); // Width that fits a shrunken path + git but not the full path + git const tightWidth = Math.floor(pathVW * 0.5) + gitVW + 10; @@ -338,3 +365,61 @@ describe("overflow: path shrinks before git is dropped", () => { } }); }); + +describe("overflow: path survives before model", () => { + it("drops the model segment before the cwd path when both cannot fit", () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "omp-statusline-overflow-")); + const cwd = path.join(root, "cwdxyz"); + fs.mkdirSync(cwd); + setProjectDir(cwd); + + const modelName = `MODEL_SHOULD_DROP_${"x".repeat(24)}`; + const session = createStatusLineSession("overflow test", modelName); + const component = new StatusLineComponent(session); + const pathOptions = { + abbreviate: false, + maxLength: 32, + stripWorkPrefix: false, + }; + component.updateSettings({ + preset: "custom", + leftSegments: ["pi", "model", "path"], + rightSegments: [], + separator: "none", + sessionAccent: false, + transparent: true, + segmentOptions: { + model: { showThinkingLevel: false }, + path: pathOptions, + }, + }); + + const ctx = { + ...createCtx({ pathMaxLength: pathOptions.maxLength }), + session, + options: { + model: { showThinkingLevel: false }, + path: pathOptions, + }, + } as SegmentContext; + const pi = renderSegment("pi", ctx).content; + const model = renderSegment("model", ctx).content; + const minPath = renderSegment("path", { + ...ctx, + options: { ...ctx.options, path: { ...pathOptions, maxLength: 4 } }, + }).content; + const separatorWidth = visibleWidth(theme.sep.space); + const groupWidth = (parts: string[]) => + parts.reduce((sum, part) => sum + visibleWidth(part), 0) + + Math.max(0, parts.length - 1) * (separatorWidth + 2) + + 2; + const width = groupWidth([pi, model]) + 1; + + expect(groupWidth([pi, model, minPath])).toBeGreaterThan(width); + expect(groupWidth([pi, minPath])).toBeLessThanOrEqual(width); + + const rendered = stripAnsi(component.getTopBorder(width).content); + expect(rendered).toContain("xyz"); + expect(rendered).not.toContain("MODEL_SHOULD_DROP"); + }); +}); From 1adc202bf5d816d769ab6199b2d6823f3a6f4ca5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 18:00:14 +0000 Subject: [PATCH 57/91] fix(coding-agent): preserved literal bash internal URLs Left unresolved internal URLs unchanged during bash command expansion so quoted literal mentions can execute verbatim. Skipped expansion for URL tokens embedded inside larger quoted shell text while preserving resolvable path-argument expansion. Fixes #4737 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/tools/bash-skill-urls.ts | 46 ++++++++++++++++--- .../test/tools/bash-skill-urls.test.ts | 44 +++++++++++------- 3 files changed, 71 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 459ec14a8..ca30fc10f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed bash internal-URL expansion so unresolved literal `memory://` / `skill://` text stays verbatim instead of aborting command execution ([#4737](https://github.com/can1357/oh-my-pi/issues/4737)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/tools/bash-skill-urls.ts b/packages/coding-agent/src/tools/bash-skill-urls.ts index 82e9e5f8e..585882a9c 100644 --- a/packages/coding-agent/src/tools/bash-skill-urls.ts +++ b/packages/coding-agent/src/tools/bash-skill-urls.ts @@ -140,6 +140,30 @@ function unquoteToken(token: string): string { return token; } +function isInsideShellQuote(command: string, index: number): boolean { + let quote: "'" | '"' | undefined; + for (let i = 0; i < index; i++) { + const char = command[i]; + if (char === "\\" && quote !== "'") { + i++; + continue; + } + if (char === "'" && quote !== '"') { + quote = quote === "'" ? undefined : "'"; + continue; + } + if (char === '"' && quote !== "'") { + quote = quote === '"' ? undefined : '"'; + } + } + return quote !== undefined; +} + +function isEmbeddedInQuotedText(command: string, token: string, index: number): boolean { + if (token.startsWith("'") || token.startsWith('"')) return false; + return isInsideShellQuote(command, index); +} + /** Shell-escape a path using single quotes. */ function shellEscape(p: string): string { return `'${p.replace(/'/g, "'\\''")}'`; @@ -216,6 +240,7 @@ export function expandSkillUrls(command: string, skills: readonly Skill[]): stri /** * Expand supported internal URLs in a bash command string to shell-escaped absolute paths. + * Unresolvable URLs and literal mentions inside larger quoted text are left unchanged. * Supported schemes: skill://, agent://, artifact://, memory://, rule://, local:// */ export async function expandInternalUrls(command: string, options: InternalUrlExpansionOptions): Promise { @@ -231,15 +256,22 @@ export async function expandInternalUrls(command: string, options: InternalUrlEx const index = match.index; if (index === undefined) continue; + if (isEmbeddedInQuotedText(command, token, index)) continue; + const rawUrl = unquoteToken(token); const url = normalizeLocalScheme(rawUrl); - const resolvedPath = await resolveInternalUrlToPath( - url, - options.skills, - options.internalRouter, - options.localOptions, - options.ensureLocalParentDirs, - ); + let resolvedPath: string; + try { + resolvedPath = await resolveInternalUrlToPath( + url, + options.skills, + options.internalRouter, + options.localOptions, + options.ensureLocalParentDirs, + ); + } catch { + continue; + } const replacement = options.noEscape ? resolvedPath : shellEscape(resolvedPath); expanded = `${expanded.slice(0, index)}${replacement}${expanded.slice(index + token.length)}`; } diff --git a/packages/coding-agent/test/tools/bash-skill-urls.test.ts b/packages/coding-agent/test/tools/bash-skill-urls.test.ts index e35fb77d4..153eef880 100644 --- a/packages/coding-agent/test/tools/bash-skill-urls.test.ts +++ b/packages/coding-agent/test/tools/bash-skill-urls.test.ts @@ -178,6 +178,22 @@ describe("expandInternalUrls", () => { ); }); + it("leaves literal internal URLs embedded in quoted text unchanged", async () => { + const router = createInternalRouter({ + "memory://root/summary.md": { sourcePath: "/tmp/memories/summary.md" }, + }); + const command = `printf '%s\\n' 'the literal memory://root/summary.md string'`; + + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); + }); + + it("leaves unresolved quoted literal URLs unchanged", async () => { + const router = createInternalRouter({}); + const command = "grep 'memory://xyz-quoted' file.txt"; + + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); + }); + it("expands agent:// URLs when router is available", async () => { const router = createInternalRouter({ "agent://abc": { sourcePath: "/tmp/session/abc.md" }, @@ -239,34 +255,30 @@ describe("expandInternalUrls", () => { ); }); - it("throws when local:// URL is used without local protocol options", async () => { - await expect(expandInternalUrls("mv foo local://bar", { skills: [] })).rejects.toThrow( - "Cannot resolve local:// URL in bash command: local protocol options are unavailable for this session.", - ); + it("leaves local:// URLs unchanged without local protocol options", async () => { + const command = "mv foo local://bar"; + await expect(expandInternalUrls(command, { skills: [] })).resolves.toBe(command); }); - it("throws when non-skill URL is used without an internal router", async () => { - await expect(expandInternalUrls("cat artifact://1", { skills: [] })).rejects.toThrow( - "Cannot resolve artifact:// URL in bash command", - ); + it("leaves non-skill URLs unchanged without an internal router", async () => { + const command = "cat artifact://1"; + await expect(expandInternalUrls(command, { skills: [] })).resolves.toBe(command); }); - it("throws when internal router resolves URL without sourcePath", async () => { + it("leaves internal URLs unchanged when they resolve without sourcePath", async () => { const router = createInternalRouter({ "rule://my-rule": {}, }); - await expect(expandInternalUrls("cat rule://my-rule", { skills: [], internalRouter: router })).rejects.toThrow( - "rule:// URL resolved without a filesystem path", - ); + const command = "cat rule://my-rule"; + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); }); - it("surfaces resolver errors with actionable context", async () => { + it("leaves internal URLs unchanged when the resolver fails", async () => { const router = createInternalRouter({ "memory://root/missing.md": { error: "Memory file not found" }, }); - await expect( - expandInternalUrls("cat memory://root/missing.md", { skills: [], internalRouter: router }), - ).rejects.toThrow("Failed to resolve memory:// URL in bash command"); + const command = "cat memory://root/missing.md"; + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); }); it("does not match local:/ inside filesystem paths (e.g. /repo/local:/PLAN.md)", async () => { From 071f199a1d923cf06250aacea163d9efc5d5f300 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 19:43:46 +0000 Subject: [PATCH 58/91] fix(agent): kept project sticky rules active Synthesized project .omp/RULES.md as a distinct sticky rule name so capability dedup no longer shadows it behind the user sticky rule. Added a public rules capability regression test covering user and project sticky RULES.md files loading together. Fixes #4739 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/discovery/builtin.ts | 3 +- .../test/discovery/builtin-rules-md.test.ts | 55 ++++++++++++++++--- 3 files changed, 53 insertions(+), 9 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 459ec14a8..2e9fbe8ed 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed project `.omp/RULES.md` sticky rules being shadowed by user `~/.omp/agent/RULES.md` rules with the same synthesized `RULES` name, so both user and project sticky rules now inject ([#4739](https://github.com/can1357/oh-my-pi/issues/4739)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/discovery/builtin.ts b/packages/coding-agent/src/discovery/builtin.ts index 94a17e2bc..42abac1d5 100644 --- a/packages/coding-agent/src/discovery/builtin.ts +++ b/packages/coding-agent/src/discovery/builtin.ts @@ -401,7 +401,8 @@ async function loadStickyRulesFile(filePath: string, level: "user" | "project"): const content = await readFile(filePath); if (!content) return null; const source = createSourceMeta(PROVIDER_ID, filePath, level); - const rule = buildRuleFromMarkdown("RULES.md", content, filePath, source, { ruleName: "RULES" }); + const ruleName = level === "project" ? "RULES@project" : "RULES"; + const rule = buildRuleFromMarkdown("RULES.md", content, filePath, source, { ruleName }); // Force alwaysApply regardless of frontmatter — the whole point of RULES.md // is to be reattached every turn. return { ...rule, alwaysApply: true }; diff --git a/packages/coding-agent/test/discovery/builtin-rules-md.test.ts b/packages/coding-agent/test/discovery/builtin-rules-md.test.ts index ff39625f5..3d7721fef 100644 --- a/packages/coding-agent/test/discovery/builtin-rules-md.test.ts +++ b/packages/coding-agent/test/discovery/builtin-rules-md.test.ts @@ -1,11 +1,9 @@ /** - * Regression test for #1266: + * Regression tests for top-level `RULES.md` sticky rules. + * * `RULES.md` (singular, top-level) MUST be loaded as a sticky always-apply rule * from both `~/.omp/agent/RULES.md` (user) and the nearest `.omp/RULES.md` * (project, walked up from cwd to repoRoot). - * - * Calls the native provider's `load` directly with the agent dir pointed at a - * tempdir (via setAgentDir) so the user scope can be staged in isolation. */ import { afterEach, beforeEach, expect, test } from "bun:test"; import * as fs from "node:fs"; @@ -15,8 +13,8 @@ import { getCapability } from "@oh-my-pi/pi-coding-agent/capability"; import { clearCache } from "@oh-my-pi/pi-coding-agent/capability/fs"; import { type Rule, ruleCapability } from "@oh-my-pi/pi-coding-agent/capability/rule"; import type { LoadContext } from "@oh-my-pi/pi-coding-agent/capability/types"; -// Register all discovery providers as a side effect. -import "@oh-my-pi/pi-coding-agent/discovery"; +// Importing discovery registers all providers as a side effect. +import { loadCapability } from "@oh-my-pi/pi-coding-agent/discovery"; import { getConfigRootDir, removeSyncWithRetries, setAgentDir } from "@oh-my-pi/pi-utils"; let tempDir: string; @@ -40,6 +38,11 @@ async function loadNativeRules(ctx: LoadContext): Promise { return result.items; } +async function loadRulesCapability(cwd: string): Promise { + const result = await loadCapability(ruleCapability.id, { cwd, providers: ["native"] }); + return result.items; +} + beforeEach(() => { clearCache(); tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-rules-md-")); @@ -81,7 +84,7 @@ test("project .omp/RULES.md becomes an alwaysApply rule", async () => { const rules = await loadNativeRules({ cwd: project, home, repoRoot: project }); - const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES"); + const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES@project"); expect(projectRule).toBeDefined(); expect(projectRule?.alwaysApply).toBe(true); expect(projectRule?.content).toContain("Always say hi."); @@ -94,12 +97,48 @@ test("project RULES.md is found walking up from a sub-package cwd", async () => const rules = await loadNativeRules({ cwd: subPkg, home, repoRoot: project }); - const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES"); + const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES@project"); expect(projectRule).toBeDefined(); expect(projectRule?.alwaysApply).toBe(true); expect(projectRule?.path).toBe(path.join(project, ".omp", "RULES.md")); }); +test("user and project sticky RULES.md both survive public capability dedup", async () => { + const userRulesPath = path.join(home, ".omp", "agent", "RULES.md"); + const projectRulesPath = path.join(project, ".omp", "RULES.md"); + const userRuleText = "User sticky rule: keep the personal safety checklist active.\n"; + const projectRuleText = "Project sticky rule: require repo-local release notes.\n"; + writeFile(userRulesPath, userRuleText); + writeFile(projectRulesPath, projectRuleText); + + const rules = await loadRulesCapability(project); + + const stickyRules = rules.filter(rule => rule.path === userRulesPath || rule.path === projectRulesPath); + expect(stickyRules).toHaveLength(2); + + const userRule = stickyRules.find(rule => rule._source.level === "user"); + const projectRule = stickyRules.find(rule => rule._source.level === "project"); + + if (!userRule) throw new Error("user sticky rule missing"); + expect(userRule.name).toBe("RULES"); + expect(userRule.path).toBe(userRulesPath); + expect(userRule._source.path).toBe(userRulesPath); + expect(userRule.alwaysApply).toBe(true); + expect(userRule.content).toContain(userRuleText.trim()); + expect("_shadowed" in userRule).toBe(false); + + if (!projectRule) throw new Error("project sticky rule missing"); + expect(projectRule.name).toBe("RULES@project"); + expect(projectRule.path).toBe(projectRulesPath); + expect(projectRule._source.path).toBe(projectRulesPath); + expect(projectRule.alwaysApply).toBe(true); + expect(projectRule.content).toContain(projectRuleText.trim()); + expect("_shadowed" in projectRule).toBe(false); + + expect(userRule.name).not.toBe(projectRule.name); + expect(userRule.content).not.toBe(projectRule.content); +}); + test("alwaysApply is forced even when frontmatter says false", async () => { writeFile(path.join(home, ".omp", "agent", "RULES.md"), "---\nalwaysApply: false\n---\nStick around anyway.\n"); From 18154fb7895a08e4a348dc590408bfd21a0d6387 Mon Sep 17 00:00:00 2001 From: Moutaz Haq Date: Thu, 2 Jul 2026 15:30:41 -0500 Subject: [PATCH 59/91] fix: Dockerfile needs to bun run gen:tool-views alongside gen:docs --- Dockerfile | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/Dockerfile b/Dockerfile index f98f81e4c..ec15e9fcc 100644 --- a/Dockerfile +++ b/Dockerfile @@ -184,9 +184,11 @@ RUN bun install --frozen-lockfile --ignore-scripts # hoisted node_modules that `bun install` just produced. COPY . /pi/ -# Regenerate the docs index that `--ignore-scripts` skipped above. The root -# package.json's `prepare` script normally handles this on a vanilla install. -RUN bun --cwd=packages/coding-agent run gen:docs +# Regenerate the docs index and tool views that `--ignore-scripts` skipped +# above. The root package.json's `prepare` script normally handles these on a +# vanilla install. +RUN bun --cwd=packages/coding-agent run gen:docs \ + && bun --cwd=/pi/packages/coding-agent run gen:tool-views ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/omp"] CMD ["--help"] From 90b1b172f1fd5084d0fbfe0aef4e8602c5b9e790 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 21:46:39 +0000 Subject: [PATCH 60/91] fix(tui): guarded selector cursor symbol - Fell back to an ASCII cursor when legacy selector themes omit the symbols map. - Added a SelectList regression test for rendering with a runtime legacy theme. Fixes #4745 --- packages/tui/CHANGELOG.md | 4 ++++ packages/tui/src/components/select-list.ts | 7 ++++--- packages/tui/test/select-list.test.ts | 10 +++++++++- 3 files changed, 17 insertions(+), 4 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index b6d4864c9..dd4c4c30e 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed selector rendering when a legacy theme omits symbol settings by falling back to an ASCII cursor instead of crashing ([#4745](https://github.com/can1357/oh-my-pi/issues/4745)). + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index 06b151a42..651de51b2 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -12,6 +12,8 @@ const DEFAULT_PRIMARY_COLUMN_WIDTH = 32; const PRIMARY_COLUMN_GAP = 2; const MIN_DESCRIPTION_WIDTH = 10; +const DEFAULT_CURSOR_SYMBOL = ">"; + function sanitizeSingleLine(text: string): string { return replaceTabs(text) .replace(/[\r\n]+/g, " ") @@ -372,9 +374,8 @@ export class SelectList implements Component, MouseRoutable { width: number, primaryColumnWidth: number, ): SelectItemLayout { - const prefix = isSelected - ? `${this.theme.symbols.cursor} ` - : padding(visibleWidth(this.theme.symbols.cursor) + 1); + const cursor = this.theme.symbols?.cursor ?? DEFAULT_CURSOR_SYMBOL; + const prefix = isSelected ? `${cursor} ` : padding(visibleWidth(cursor) + 1); const prefixWidth = visibleWidth(prefix); const descriptionSingleLine = item.description ? sanitizeSingleLine(item.description) : undefined; diff --git a/packages/tui/test/select-list.test.ts b/packages/tui/test/select-list.test.ts index f9b2128ed..4d5eefe78 100644 --- a/packages/tui/test/select-list.test.ts +++ b/packages/tui/test/select-list.test.ts @@ -1,5 +1,5 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { SelectList } from "@oh-my-pi/pi-tui/components/select-list"; +import { SelectList, type SelectListTheme } from "@oh-my-pi/pi-tui/components/select-list"; import { KeybindingsManager, setKeybindings, TUI_KEYBINDINGS } from "@oh-my-pi/pi-tui/keybindings"; import type { SgrMouseEvent } from "@oh-my-pi/pi-tui/mouse"; import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; @@ -78,6 +78,14 @@ describe("SelectList", () => { expect(rendered[0]).toContain("Line one Line two Line three"); }); + it("falls back to an ASCII cursor when a legacy theme omits symbols", () => { + const legacyTheme: SelectListTheme = { ...testTheme }; + Reflect.deleteProperty(legacyTheme, "symbols"); + const list = new SelectList([{ value: "run", label: "run" }], 1, legacyTheme); + + expect(list.render(40)).toEqual(["> run"]); + }); + it("keeps descriptions aligned when the primary text is truncated", () => { const items = [ { value: "short", label: "short", description: "short description" }, From 8aae263eafa6ea33fee0881dc6c70613b22b5604 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 22:23:14 +0000 Subject: [PATCH 61/91] fix(catalog): preserved litellm vision metadata - Continued LiteLLM rich discovery past /model_group/info when vision metadata is missing. - Merged later /model/info capability metadata without losing earlier display names. - Added regression coverage for model_info.supports_vision from LiteLLM proxies. Fixes #4747 --- packages/catalog/CHANGELOG.md | 4 + .../src/provider-models/openai-compat.ts | 70 +++++++++++++--- .../catalog/test/litellm-provider.test.ts | 82 +++++++++++++++++-- 3 files changed, 138 insertions(+), 18 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 42e39218a..1de902086 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed LiteLLM discovery stopping at `/model_group/info` when that endpoint omitted `supports_vision`; it now continues to `/model/info` and preserves `model_info.supports_vision=true` for vision-capable proxy models. ([#4747](https://github.com/can1357/oh-my-pi/issues/4747)) + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index d43a20143..87ed0c95a 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3032,6 +3032,11 @@ export interface FetchLiteLLMRichModelsOptions { } type LiteLLMRichModelEntry = Record; +type LiteLLMRichEndpointModel = { + model: ModelSpec; + supportsVision: unknown; + supportsReasoning: unknown; +}; const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const; export const OPENAI_COMPAT_DISCOVERY_DEFAULT_CONTEXT_WINDOW = 128_000; @@ -3264,7 +3269,7 @@ async function fetchLiteLLMRichEndpoint( managementBaseUrl: string, runtimeBaseUrl: string, signal?: AbortSignal, -): Promise[] | null> { +): Promise<{ models: LiteLLMRichEndpointModel[]; incompleteVisionMetadata: boolean } | null> { const fetchImpl = discoveryFetch(options.fetch); const requestHeaders: Record = { Accept: "application/json", @@ -3296,17 +3301,29 @@ async function fetchLiteLLMRichEndpoint( if (!entries || entries.length === 0) { return null; } - const deduped = new Map>(); + const deduped = new Map>(); + let incompleteVisionMetadata = false; for (const entry of entries) { const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl); if (model) { - deduped.set(model.id, model); + const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision"); + if (supportsVision !== true && supportsVision !== false) { + incompleteVisionMetadata = true; + } + deduped.set(model.id, { + model, + supportsVision, + supportsReasoning: getLiteLLMMetadataValue(entry, "supports_reasoning"), + }); } } if (deduped.size === 0) { return null; } - return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id)); + return { + models: Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id)), + incompleteVisionMetadata, + }; } export async function fetchLiteLLMRichModels( @@ -3318,13 +3335,40 @@ export async function fetchLiteLLMRichModels( return null; } const fetchModels = async (signal?: AbortSignal): Promise[] | null> => { + const deduped = new Map>(); for (const endpoint of LITELLM_RICH_ENDPOINTS) { - const models = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); - if (models) { - return models; + const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); + if (!result) { + continue; + } + for (const next of result.models) { + const existing = deduped.get(next.model.id); + if (!existing) { + deduped.set(next.model.id, next); + continue; + } + const model = { + ...existing.model, + ...next.model, + name: next.model.name === next.model.id ? existing.model.name : next.model.name, + input: + next.supportsVision === true || next.supportsVision === false + ? next.model.input + : existing.model.input, + reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning, + }; + deduped.set(next.model.id, { ...next, model }); + } + if (!result.incompleteVisionMetadata) { + break; } } - return null; + if (deduped.size === 0) { + return null; + } + return Array.from(deduped.values()) + .map(entry => entry.model) + .sort((left, right) => left.id.localeCompare(right.id)); }; if (options.signal !== undefined) { return fetchModels(options.signal); @@ -3339,11 +3383,11 @@ export function litellmModelManagerOptions( const baseUrl = config?.baseUrl ?? Bun.env.LITELLM_BASE_URL ?? "http://localhost:4000/v1"; return { providerId: "litellm", - // rich-v3 invalidates rows cached before reseller usage-suffix stripping - // and placeholder-only `all-team-models` filtering; bump the version whenever - // the mappers below change, or warm authoritative caches keep serving - // pre-change rows for the full TTL. - cacheProviderId: `litellm:rich-v3:${Bun.hash(baseUrl).toString(36)}`, + // rich-v4 invalidates rows cached before discovery continued past + // `/model_group/info` when that endpoint omitted vision metadata; bump the + // version whenever the mappers below change, or warm authoritative caches keep + // serving pre-change rows for the full TTL. + cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy and is never bundled in models.json (that // would leak the machine's localhost catalog). Prefer the proxy's richer // management metadata, then fall back to /v1/models and enrich bare ids diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index ec6d88704..5dc72ff23 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -125,7 +125,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v3:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, + `litellm:rich-v4:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -148,7 +148,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v3:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, + `litellm:rich-v4:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -247,8 +247,13 @@ describe("LiteLLM provider discovery", () => { if (url === "http://primary:4000/model_group/info") { return Response.json({ data: [ - { model_group: "no-tools", providers: ["openai"], supports_function_calling: false }, - { model_group: "params-tools", supported_openai_params: ["tools"] }, + { + model_group: "no-tools", + providers: ["openai"], + supports_vision: false, + supports_function_calling: false, + }, + { model_group: "params-tools", supports_vision: false, supported_openai_params: ["tools"] }, ], }); } @@ -339,6 +344,7 @@ describe("LiteLLM provider discovery", () => { max_input_tokens: 96_000, max_output_tokens: 8_000, supports_function_calling: true, + supports_vision: false, }, ], }); @@ -422,6 +428,70 @@ describe("LiteLLM provider discovery", () => { }); }); + test("continues to LiteLLM model info when model_group omits vision metadata", async () => { + const calls: string[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + calls.push(url); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ + data: [ + { + model_group: "vision-proxy-model", + model_name: "Vision Proxy Model", + max_input_tokens: 128_000, + max_output_tokens: 16_000, + }, + ], + }); + } + if (url === "http://primary:4000/v2/model/info") { + return new Response("{}", { status: 404 }); + } + if (url === "http://primary:4000/model/info") { + return Response.json({ + data: [ + { model_name: "text-only-model", model_info: { supports_vision: false } }, + { + model_name: "vision-proxy-model", + model_info: { + max_input_tokens: 128_000, + max_output_tokens: 16_000, + supports_vision: true, + }, + }, + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when LiteLLM model info succeeds"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(calls).toContain("http://primary:4000/model_group/info"); + expect(calls).toContain("http://primary:4000/v2/model/info"); + expect(calls).toContain("http://primary:4000/model/info"); + expect(calls).not.toContain("http://primary:4000/v1/models"); + expect(models?.find(model => model.id === "vision-proxy-model")).toMatchObject({ + id: "vision-proxy-model", + name: "Vision Proxy Model", + input: ["text", "image"], + contextWindow: 128_000, + maxTokens: 16_000, + }); + }); + test("falls back from v2 model info to LiteLLM model info", async () => { const calls: string[] = []; const fetchMock = vi.fn(async (input: string | URL | Request) => { @@ -431,7 +501,9 @@ describe("LiteLLM provider discovery", () => { return new Response("{}", { status: 404 }); } if (url === "http://primary:4000/model/info") { - return Response.json({ data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000 } }] }); + return Response.json({ + data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000, supports_vision: false } }], + }); } throw new Error(`Unexpected URL: ${url}`); }) as FetchImpl; From e812c368ca67c493acf9c001cbe9255ea9f59cf5 Mon Sep 17 00:00:00 2001 From: vmcall Date: Tue, 7 Jul 2026 17:04:34 +0200 Subject: [PATCH 62/91] fix(plan): carried local artifacts into execution --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/modes/interactive-mode.ts | 51 ++++++++++++++++- .../src/prompts/system/plan-mode-active.md | 7 ++- .../test/interactive-mode-plan-review.test.ts | 55 +++++++++++++++++++ 4 files changed, 112 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..21c96f4c9 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed plan mode to document `local://` artifacts as writable session-local planning files and to carry every pre-approval `local://` artifact into the fresh session created by Approve and Execute. + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8e064d8ee..ee4414604 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -2602,6 +2602,49 @@ export class InteractiveMode implements InteractiveModeContext { } } + #resolveLocalRoot(): string { + return resolveLocalUrlToPath("local://", { + getArtifactsDir: () => this.sessionManager.getArtifactsDir(), + getSessionId: () => this.sessionManager.getSessionId(), + }); + } + + async #copyLocalArtifactsForFreshSession(sourceRoot: string, destinationRoot: string): Promise { + if (sourceRoot === destinationRoot) return; + + let sourceRootStat: { isDirectory(): boolean }; + try { + sourceRootStat = await fs.lstat(sourceRoot); + } catch (error) { + if (isEnoent(error)) return; + throw error; + } + + if (!sourceRootStat.isDirectory()) return; + + await fs.mkdir(destinationRoot, { recursive: true }); + await this.#copyLocalArtifactEntries(sourceRoot, destinationRoot); + } + + async #copyLocalArtifactEntries(sourceDir: string, destinationDir: string): Promise { + const entries = await fs.readdir(sourceDir, { withFileTypes: true }); + for (const entry of entries) { + const sourcePath = path.join(sourceDir, entry.name); + const destinationPath = path.join(destinationDir, entry.name); + + if (entry.isDirectory()) { + await fs.mkdir(destinationPath, { recursive: true }); + await this.#copyLocalArtifactEntries(sourcePath, destinationPath); + continue; + } + + if (entry.isFile()) { + await fs.mkdir(path.dirname(destinationPath), { recursive: true }); + await fs.copyFile(sourcePath, destinationPath); + } + } + } + async #approvePlan( planContent: string, options: { @@ -2632,14 +2675,16 @@ export class InteractiveMode implements InteractiveModeContext { }); if (!options.preserveContext) { + const oldLocalRoot = this.#resolveLocalRoot(); await this.handleClearCommand(); - // The new session has a fresh local:// root — persist the approved plan there - // so `local://-plan.md` resolves correctly in the execution session. + const newLocalRoot = this.#resolveLocalRoot(); + await this.#copyLocalArtifactsForFreshSession(oldLocalRoot, newLocalRoot); const newLocalPath = resolveLocalUrlToPath(options.planFilePath, { getArtifactsDir: () => this.sessionManager.getArtifactsDir(), getSessionId: () => this.sessionManager.getSessionId(), }); - await Bun.write(newLocalPath, planContent); + await fs.mkdir(path.dirname(newLocalPath), { recursive: true }); + await fs.writeFile(newLocalPath, planContent); } else if (options.compactBeforeExecute) { // Distill the plan-mode transcript before the execution turn is queued so // the plan-approved synthetic prompt lands as a fresh cache anchor. diff --git a/packages/coding-agent/src/prompts/system/plan-mode-active.md b/packages/coding-agent/src/prompts/system/plan-mode-active.md index dcb9bce64..4f56ba5a9 100644 --- a/packages/coding-agent/src/prompts/system/plan-mode-active.md +++ b/packages/coding-agent/src/prompts/system/plan-mode-active.md @@ -1,7 +1,10 @@ -Plan mode is active. You MUST perform READ-ONLY work only: -- You NEVER create, edit, or delete files — except the single plan file named below. +Plan mode is active. You MUST preserve read-only working-tree and system semantics: +- You NEVER create, edit, delete, or rename working-tree files. - You NEVER run state-changing commands (`git commit`, `npm install`, migrations) or make any other system change. +- `local://` artifacts are session-local planning artifacts. You MAY create or update them when explicitly requested or needed for the plan. +- You NEVER delete or rename `local://` artifacts. +- You MUST write the canonical plan to `local://-plan.md`. To leave plan mode and implement: call `resolve` with `action: "apply"`, a `reason`, and `extra: { title: "" }`, where `` matches your `local://-plan.md`. The user then picks an execution option and full write access is restored. `` may contain only letters, numbers, underscores, and hyphens. diff --git a/packages/coding-agent/test/interactive-mode-plan-review.test.ts b/packages/coding-agent/test/interactive-mode-plan-review.test.ts index 7efce822d..6ebf4a59a 100644 --- a/packages/coding-agent/test/interactive-mode-plan-review.test.ts +++ b/packages/coding-agent/test/interactive-mode-plan-review.test.ts @@ -415,6 +415,61 @@ describe("InteractiveMode plan review rendering", () => { expect(await Bun.file(resolvedPlanPath).text()).toContain("edited body"); }); + it("carries pre-approval local artifacts into the fresh approve-and-execute session", async () => { + const planFilePath = "local://handoff-plan.md"; + const localOptions = { + getArtifactsDir: () => session.sessionManager.getArtifactsDir(), + getSessionId: () => session.sessionManager.getSessionId(), + }; + const oldLocalRoot = resolveLocalUrlToPath("local://", localOptions); + const oldPlanPath = resolveLocalUrlToPath(planFilePath, localOptions); + const oldArtifactPath = resolveLocalUrlToPath("local://handoff/nested/context.txt", localOptions); + await fs.mkdir(path.dirname(oldArtifactPath), { recursive: true }); + await Bun.write(oldArtifactPath, "pre-approval handoff"); + await Bun.write(oldPlanPath, "# Plan\n\noriginal body\n"); + + mode.planModeEnabled = true; + mode.planModePlanFilePath = planFilePath; + const planContent = "# Plan\n\nfinal approved body\n"; + vi.spyOn(mode, "showPlanReview").mockImplementation(async (_plan, _title, _options, dialogOptions) => { + dialogOptions?.onPlanEdited?.(planContent); + return "Approve and execute"; + }); + vi.spyOn(mode, "handleClearCommand").mockImplementation(async () => { + await session.sessionManager.newSession(); + }); + let artifactAtPrompt = ""; + let planAtPrompt = ""; + const prompt = vi.spyOn(session, "prompt").mockImplementation(async () => { + const promptArtifactPath = resolveLocalUrlToPath("local://handoff/nested/context.txt", localOptions); + const promptPlanPath = resolveLocalUrlToPath(planFilePath, localOptions); + artifactAtPrompt = (await Bun.file(promptArtifactPath).exists()) + ? await Bun.file(promptArtifactPath).text() + : ""; + planAtPrompt = (await Bun.file(promptPlanPath).exists()) ? await Bun.file(promptPlanPath).text() : ""; + return undefined as never; + }); + + expect(await Bun.file(oldArtifactPath).text()).toBe("pre-approval handoff"); + + await mode.handlePlanApproval({ + planFilePath, + planExists: true, + title: "HANDOFF", + }); + + const newLocalRoot = resolveLocalUrlToPath("local://", localOptions); + const newArtifactPath = resolveLocalUrlToPath("local://handoff/nested/context.txt", localOptions); + const newPlanPath = resolveLocalUrlToPath(planFilePath, localOptions); + expect(newLocalRoot).not.toBe(oldLocalRoot); + expect(await Bun.file(newArtifactPath).text()).toBe("pre-approval handoff"); + expect(await Bun.file(newPlanPath).text()).toBe(planContent); + expect(artifactAtPrompt).toBe("pre-approval handoff"); + expect(planAtPrompt).toBe(planContent); + expect(await Bun.file(oldArtifactPath).text()).toBe("pre-approval handoff"); + expect(prompt).toHaveBeenCalledWith(expect.any(String), { synthetic: true }); + }); + it("offers approve-and-keep-context as a distinct plan approval path", async () => { const planFilePath = "local://PLAN.md"; const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, { From d23e3296c1fe312940200e2bebb52a40ac3c2911 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Tue, 7 Jul 2026 10:46:12 -0300 Subject: [PATCH 63/91] fix(ai): kept assistant tool_use blocks trailing on Anthropic replay Anthropic's replay validator rejects any non-tool_use block after a tool_use inside an assistant message. A mid-turn server-side fallback handoff persisted after a completed tool call (fallback block + continued text/tool calls, persisted since #4182) replayed verbatim and 400ed every subsequent request, wedging the session. Stable-partition assistant wire blocks in convertAnthropicMessages: non-tool_use blocks first in original order (preserving the thinking-signature/fallback chain), tool_use blocks trailing, with an identity fast-path so already-valid turns stay byte-identical for prompt caching. Same family as the cross-provider text-after-tool_use shape in #544. Fixes #4781 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/anthropic.ts | 34 +++ .../anthropic-server-side-fallback.test.ts | 255 +++++++++++++++++- 3 files changed, 292 insertions(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9c99508fd..1bdba4369 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Anthropic replay 400s (`tool_use ids were found without tool_result blocks immediately after`) when a persisted assistant turn carries content after a completed tool call — such as a mid-turn `server-side-fallback` handoff (fallback block plus continued text/tool calls after the primary model's `tool_use`) or trailing text from cross-provider replays — by stable-partitioning assistant content so all `tool_use` blocks trail the non-`tool_use` chain. ([#4781](https://github.com/can1357/oh-my-pi/issues/4781), [#544](https://github.com/can1357/oh-my-pi/issues/544)) + ## [16.3.11] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 5093c7795..a4df52aa0 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -3529,6 +3529,40 @@ export function convertAnthropicMessages( }); } } + // Anthropic's replay validator rejects any non-`tool_use` block that + // appears after a `tool_use` inside an assistant turn (400: + // "tool_use ids were found without tool_result blocks immediately + // after: "). A persisted turn can violate this when a mid-turn + // server-side fallback handoff lands after the primary model already + // emitted a tool_use — the replayed content is then e.g. + // [thinking, text, tool_use, fallback, text, tool_use] — and also for + // the older cross-provider [text, tool_use, text] shape (issue #544). + // Stable-partition into [...non-tool_use, ...tool_use], preserving each + // side's relative order: the non-tool_use chain (thinking → text → + // fallback → text) carries thinking signatures and the fallback + // boundary marker whose order Anthropic verifies, while tool_use blocks + // are unsigned and safe to defer to the tail. Fast-path untouched when + // already in order so prompt-cache prefixes stay byte-identical. + let sawToolUse = false; + let needsPartition = false; + for (const block of blocks) { + if (block.type === "tool_use") { + sawToolUse = true; + } else if (sawToolUse) { + needsPartition = true; + break; + } + } + if (needsPartition) { + const nonToolUse: ContentBlockParam[] = []; + const toolUse: ContentBlockParam[] = []; + for (const block of blocks) { + if (block.type === "tool_use") toolUse.push(block); + else nonToolUse.push(block); + } + blocks.length = 0; + blocks.push(...nonToolUse, ...toolUse); + } if (blocks.length === 0) continue; params.push({ role: "assistant", diff --git a/packages/ai/test/anthropic-server-side-fallback.test.ts b/packages/ai/test/anthropic-server-side-fallback.test.ts index 9a1beec8c..5cae3490e 100644 --- a/packages/ai/test/anthropic-server-side-fallback.test.ts +++ b/packages/ai/test/anthropic-server-side-fallback.test.ts @@ -17,7 +17,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import { AnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic-client"; -import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; +import type { MessageParam } from "@oh-my-pi/pi-ai/providers/anthropic-wire"; +import type { AssistantMessage, Context, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; const fableModel: Model<"anthropic-messages"> = buildModel({ @@ -328,3 +329,255 @@ describe("anthropic fallback content-block replay policy", () => { expect(params[0]?.content).toEqual([{ type: "text", text: "continued" }]); }); }); + +describe("anthropic assistant replay block ordering (tool_use partition)", () => { + // Anthropic's replay validator rejects any assistant turn that carries a + // non-`tool_use` content block AFTER a `tool_use` block + // (`messages.N: tool_use ids were found without tool_result blocks + // immediately after`). This bites when a mid-turn server-side fallback + // (`server-side-fallback-2026-06-01`) lands AFTER the primary model already + // emitted a tool_use, leaving persisted content shaped like + // [thinking, text, tool_use, fallback, text, tool_use]. The same validator + // rejects the older cross-provider [text, tool_use, text] shape (issue + // #544). `convertAnthropicMessages` defends against both by stable- + // partitioning every assistant wire message: all non-tool_use blocks first + // (original relative order), then all tool_use blocks (original relative + // order). Already-valid messages must serialize byte-identically. + + function assistant(content: AssistantMessage["content"]): AssistantMessage { + return { + role: "assistant", + content, + api: "anthropic-messages", + provider: "anthropic", + model: "claude-fable-5", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 0, + }; + } + + function toolResult(id: string, name: string): ToolResultMessage { + return { + role: "toolResult", + toolCallId: id, + toolName: name, + content: [{ type: "text", text: "ok" }], + isError: false, + timestamp: 0, + }; + } + + function assistantParam(params: MessageParam[]): MessageParam { + const asst = params.filter(p => p.role === "assistant"); + if (asst.length !== 1 || !Array.isArray(asst[0]?.content)) { + throw new Error("expected exactly one assistant param with structured content blocks"); + } + return asst[0]; + } + + it("mid-turn fallback repro: defers trailing tool_use, keeps the fallback marker in place", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "after" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + { type: "toolCall", id: "call_c", name: "glob", arguments: {} }, + ]), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + toolResult("call_c", "glob"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "plan", signature: "sig-1" }, + { type: "text", text: "before" }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "after" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + { type: "tool_use", id: "call_b", name: "grep", input: {} }, + { type: "tool_use", id: "call_c", name: "glob", input: {} }, + ]); + }); + + it("opt-out drops the fallback marker but still defers tool_use to the tail", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "after" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + { type: "toolCall", id: "call_c", name: "glob", arguments: {} }, + ]), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + toolResult("call_c", "glob"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + // serverSideFallbackEnabled omitted → fallback block dropped + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "plan", signature: "sig-1" }, + { type: "text", text: "before" }, + { type: "text", text: "after" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + { type: "tool_use", id: "call_b", name: "grep", input: {} }, + { type: "tool_use", id: "call_c", name: "glob", input: {} }, + ]); + }); + + it("issue-544 family: text after a tool_use is reordered before the tool_use", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "text", text: "after" }, + ]), + toolResult("call_a", "read"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "text", text: "before" }, + { type: "text", text: "after" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + ]); + }); + + it("identity fast-path: already-valid thinking→text→tool_use serializes in unchanged order", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + ]), + toolResult("call_a", "read"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "plan", signature: "sig-1" }, + { type: "text", text: "before" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + ]); + }); + + it("interleaved signed thinking: signature-chain order preserved, tool_use deferred to the tail", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "first", thinkingSignature: "sig-1" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "thinking", thinking: "second", thinkingSignature: "sig-2" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + ]), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "first", signature: "sig-1" }, + { type: "thinking", thinking: "second", signature: "sig-2" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + { type: "tool_use", id: "call_b", name: "grep", input: {} }, + ]); + }); + + it("partition is localized: all other wire messages serialize byte-identically", () => { + // Prompt-cache contract: partitioning a poisoned assistant turn must be a + // LOCAL rewrite of that turn's own content — every OTHER wire message + // (the earlier valid assistant turn whose cached prefix must survive, its + // tool_result, the poisoned turn's tool_results, and the trailing user + // turn) must serialize to the exact same bytes. If the reorder leaked past + // the turn boundary (mutated a shared/adjacent message) or the slow path + // diverged from an already-ordered fast path, cached prefixes up to the + // reordered turn would be invalidated. Two histories, identical except the + // poisoned turn is pre-ordered to the partition result in (b): (a) fires + // the partition, (b) takes the fast path. Bytes must match everywhere but + // the poisoned param, which must converge to the same content either way. + const validAssistant = assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_v", name: "list", arguments: {} }, + ]); + const poisonedContent: AssistantMessage["content"] = [ + { type: "text", text: "poison-a" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "poison-b" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + ]; + // Same blocks, hand-ordered to exactly what the stable partition emits: + // non-tool_use chain first (order preserved), then the tool_use tail. + const preOrderedContent: AssistantMessage["content"] = [ + { type: "text", text: "poison-a" }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "poison-b" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + ]; + const history = (poisoned: AssistantMessage["content"]) => [ + { role: "user", content: "start", timestamp: 0 } as const, + validAssistant, + toolResult("call_v", "list"), + assistant(poisoned), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + { role: "user", content: "next", timestamp: 0 } as const, + ]; + + const a = convertAnthropicMessages(history(poisonedContent), fableModel, false, { + serverSideFallbackEnabled: true, + }); + const b = convertAnthropicMessages(history(preOrderedContent), fableModel, false, { + serverSideFallbackEnabled: true, + }); + + expect(a.length).toBe(b.length); + const poisonedIdx = a.findIndex( + p => + p.role === "assistant" && + Array.isArray(p.content) && + p.content.some(block => block.type === "tool_use" && block.id === "call_a"), + ); + expect(poisonedIdx).toBeGreaterThanOrEqual(0); + for (let i = 0; i < a.length; i++) { + if (i === poisonedIdx) continue; + expect(JSON.stringify(a[i])).toBe(JSON.stringify(b[i])); + } + // Convergence: partition (a) and fast path (b) yield identical final content. + expect(a[poisonedIdx]).toEqual(b[poisonedIdx]); + }); +}); From ed4b5b5b0a62be18c03f484de7ce09972705e49f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Tue, 7 Jul 2026 13:27:11 -0300 Subject: [PATCH 64/91] fix(coding-agent): resolved agent:// for nested subagent output artifactsDirsFromRegistry scanned only each registered agent's adopted (root-wide) ArtifactManager dir, but a subagent's own children are written one level deeper under its sessionFile-derived dir (task/index.ts). Any spawn chain 2+ levels deep therefore had its live, addressable output unresolvable via agent:// and the output() eval helper (Not found / Available: none). Collect both candidate dirs per ref; addDir dedup collapses the depth-0 case. Fixes #4650 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../__tests__/agent-protocol-nested.test.ts | 68 +++++++++++++++++++ .../src/internal-urls/registry-helpers.ts | 15 ++-- 3 files changed, 81 insertions(+), 6 deletions(-) create mode 100644 packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..c882f6b27 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed `agent://` (and the `output()` eval helper) failing with `Not found` for a subagent spawned by another subagent (any spawn chain 2+ levels deep). `artifactsDirsFromRegistry` scanned only each ref's adopted (root-wide) `ArtifactManager` dir, but a subagent's own children are written one level deeper under its `sessionFile`-derived dir — so a live, addressable nested peer's output was unresolvable. The resolver now collects both candidate dirs per registered agent. ([#4650](https://github.com/can1357/oh-my-pi/issues/4650)) + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts b/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts new file mode 100644 index 000000000..7829557f5 --- /dev/null +++ b/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts @@ -0,0 +1,68 @@ +import { afterAll, afterEach, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { TempDir } from "@oh-my-pi/pi-utils"; +import { AgentRegistry } from "../../registry/agent-registry"; +import type { AgentSession } from "../../session/agent-session"; +import { ArtifactManager } from "../../session/artifacts"; +import { AgentProtocolHandler } from "../agent-protocol"; +import { resetRegisteredArtifactDirsForTests } from "../registry-helpers"; + +const tempDir = TempDir.createSync("omp-nested-agent-repro-"); +afterEach(() => { + AgentRegistry.resetGlobalForTests(); + resetRegisteredArtifactDirsForTests(); +}); +afterAll(() => { + tempDir.removeSync(); +}); + +it("agent:// resolves a depth-2 subagent's .md output while its session is live and artifact-manager-adopted", async () => { + const root = tempDir.path(); + const rootSessionFile = path.join(root, "session.jsonl"); + const rootArtifactsDir = rootSessionFile.slice(0, -6); + await fs.mkdir(rootArtifactsDir, { recursive: true }); + // Every subagent adopts the root ArtifactManager and reports its dir. + const sharedArtifactManager = new ArtifactManager(rootArtifactsDir); + + // A depth-1 subagent's OWN children are written under its own + // sessionFile.slice(0, -6) (task/index.ts), i.e. one level deeper. + const midSessionFile = path.join(rootArtifactsDir, "CodexDeepDive.jsonl"); + const midOwnArtifactsDir = midSessionFile.slice(0, -6); + await fs.mkdir(midOwnArtifactsDir, { recursive: true }); + + const grandchildId = "CodexDeepDive.GraphStore"; + const grandchildSessionFile = path.join(midOwnArtifactsDir, `${grandchildId}.jsonl`); + await fs.writeFile(path.join(midOwnArtifactsDir, `${grandchildId}.md`), "full report content"); + + const fakeSession = { + sessionManager: { getArtifactsDir: () => sharedArtifactManager.dir }, + } as unknown as AgentSession; + const registry = AgentRegistry.global(); + registry.register({ + id: "Main", + displayName: "main", + kind: "main", + session: fakeSession, + sessionFile: rootSessionFile, + }); + registry.register({ + id: "CodexDeepDive", + displayName: "sub", + kind: "sub", + parentId: "Main", + session: fakeSession, + sessionFile: midSessionFile, + }); + registry.register({ + id: grandchildId, + displayName: "sub", + kind: "sub", + parentId: "CodexDeepDive", + session: fakeSession, + sessionFile: grandchildSessionFile, + }); + + const resource = await new AgentProtocolHandler().resolve(new URL(`agent://${grandchildId}`) as never); + expect(resource.content).toBe("full report content"); +}); diff --git a/packages/coding-agent/src/internal-urls/registry-helpers.ts b/packages/coding-agent/src/internal-urls/registry-helpers.ts index 02648f76a..701ed685f 100644 --- a/packages/coding-agent/src/internal-urls/registry-helpers.ts +++ b/packages/coding-agent/src/internal-urls/registry-helpers.ts @@ -20,11 +20,13 @@ export function resetRegisteredArtifactDirsForTests(): void { /** * Snapshot of artifacts dirs for every registered session, deduped. * - * Prefers `sessionManager.getArtifactsDir()` because subagents adopt their - * parent's `ArtifactManager` and report the parent's dir there; dedup then - * collapses parent + N subagents (the whole agent tree) to one entry. Falls - * back to the raw session file (with the `.jsonl` suffix stripped) when no - * live session reference is attached. + * Collects TWO candidate dirs per ref, because a subagent reads from its + * adopted (root-wide) `ArtifactManager.dir` but its own children are written + * one level deeper, under `sessionFile.slice(0, -6)` (`task/index.ts`). A + * depth-2+ subagent's output therefore lives in the write-time dir, not the + * adopted one, so `agent://` must scan both or it 404s a live nested peer. + * `addDir` dedup collapses the depth-0 case (both formulas agree) back to a + * single entry. */ export function artifactsDirsFromRegistry(): string[] { const dirs: string[] = []; @@ -33,7 +35,8 @@ export function artifactsDirsFromRegistry(): string[] { if (!dirs.includes(dir)) dirs.push(dir); }; for (const ref of AgentRegistry.global().list()) { - addDir(ref.session?.sessionManager.getArtifactsDir() ?? (ref.sessionFile ? ref.sessionFile.slice(0, -6) : null)); + addDir(ref.session?.sessionManager.getArtifactsDir()); + if (ref.sessionFile) addDir(ref.sessionFile.slice(0, -6)); } for (const dir of extraArtifactsDirs) addDir(dir); return dirs; From 1a6065bebc4d4d7519c3e8be93266364248a3311 Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Wed, 8 Jul 2026 00:22:55 +0800 Subject: [PATCH 65/91] fix(auth): prefer sibling credentials before provider fallback --- packages/ai/CHANGELOG.md | 16 ++++ packages/ai/src/auth-gateway/server.ts | 1 + packages/ai/src/auth-storage.ts | 14 +++- .../test/agent-session-retry-cap.test.ts | 73 +++++++++++++++++++ .../test/auth-storage-rotation.test.ts | 24 ++++++ 5 files changed, 126 insertions(+), 2 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9c99508fd..06c456631 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,22 @@ ## [Unreleased] +### Fixed + +- Fixed access-token-only OAuth credentials attempting token refresh with an empty refresh token after expiry. +- Fixed gateway usage-limit retries falling through to cross-provider model fallback before trying a sibling credential from the same provider. + +## [16.3.11-zen.1] - 2026-07-07 + +### Added + +- Added `anysearch` registry provider definition and interactive credentials login support. + +### Fixed + +- Fixed plain-text 5xx provider status messages (including relay `520` HTML pages) being left as non-retryable numeric statuses instead of transient retryable errors. +- Fixed relay `insufficient_user_*` quota errors being classified as generic 403 auth failures instead of retryable usage-limit errors. + ## [16.3.11] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index 3caf4808e..c7eff02f4 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -233,6 +233,7 @@ async function refreshGatewayApiKeyAfterAuthError( retryAfterMs, baseUrl: model.baseUrl, modelId: model.id, + apiKey: oldKey, signal, }); logger.debug("auth-gateway retrying provider request after usage-limit block", { diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index a790ce5d5..38c896e7e 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3057,9 +3057,19 @@ export class AuthStorage { async markUsageLimitReached( provider: string, sessionId: string | undefined, - options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; signal?: AbortSignal }, + options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; apiKey?: string; signal?: AbortSignal }, ): Promise { - const sessionCredential = this.#getSessionCredential(provider, sessionId); + let sessionCredential = this.#getSessionCredential(provider, sessionId); + if (!sessionCredential && options?.apiKey) { + const stored = this.#getStoredCredentials(provider); + for (let index = 0; index < stored.length; index++) { + const entry = stored[index]; + if (entry && (await this.#credentialMatchesApiKey(entry.credential, options.apiKey))) { + sessionCredential = { type: entry.credential.type, index }; + break; + } + } + } if (!sessionCredential) return { switched: false }; const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type); diff --git a/packages/coding-agent/test/agent-session-retry-cap.test.ts b/packages/coding-agent/test/agent-session-retry-cap.test.ts index d9b8d1333..1efa5c1bd 100644 --- a/packages/coding-agent/test/agent-session-retry-cap.test.ts +++ b/packages/coding-agent/test/agent-session-retry-cap.test.ts @@ -282,6 +282,79 @@ describe("AgentSession retry delay cap", () => { expect(last.content).toContainEqual({ type: "text", text: "recovered after credential switch" }); }); + it("switches same-provider credentials before model fallback on ChatGPT usage limits", async () => { + const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + const fallbackModel = getBundledModel("openai", "gpt-5.5"); + if (!primaryModel || !fallbackModel) { + throw new Error("Expected bundled primary and fallback test models to exist"); + } + + authStorage.removeRuntimeApiKey("anthropic"); + authStorage.setRuntimeApiKey("openai", "openai-fallback-key"); + await authStorage.set("anthropic", [ + { type: "api_key", key: "anthropic-key-1" }, + { type: "api_key", key: "anthropic-key-2" }, + ]); + + const usageLimitError = "Error: You have hit your ChatGPT usage limit (k12 plan). Try again in ~231 min."; + const mock = createMockModel(); + const requestedModels: string[] = []; + const requestedKeys: string[] = []; + let agent!: Agent; + agent = new Agent({ + getApiKey: model => modelRegistry.resolver(model, agent.sessionId), + initialState: { + model: primaryModel, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + streamFn: (requestedModel, context, options) => { + requestedModels.push(`${requestedModel.provider}/${requestedModel.id}`); + const apiKey = resolveInitialApiKey(options?.apiKey); + requestedKeys.push(apiKey); + if (requestedKeys.length === 1) { + mock.push({ throw: usageLimitError }); + } else { + mock.push({ content: ["recovered after sibling account"] }); + } + return mock.stream(requestedModel, context, options); + }, + }); + + const settings = Settings.isolated({ + "compaction.enabled": false, + "retry.baseDelayMs": 5, + "retry.maxDelayMs": 100, + "retry.maxRetries": 1, + "retry.modelFallback": true, + "retry.fallbackChains": { + default: [`${fallbackModel.provider}/${fallbackModel.id}`], + }, + }); + settings.setModelRole("default", `${primaryModel.provider}/${primaryModel.id}`); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + + vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); + await session.prompt("Trigger k12 usage limit"); + await session.waitForIdle(); + + expect(requestedModels).toEqual([ + `${primaryModel.provider}/${primaryModel.id}`, + `${primaryModel.provider}/${primaryModel.id}`, + ]); + expect([...requestedKeys].sort()).toEqual(["anthropic-key-1", "anthropic-key-2"]); + const last = lastAssistant(session); + expect(last.stopReason).toBe("stop"); + expect(last.content).toContainEqual({ type: "text", text: "recovered after sibling account" }); + }); + it("waits for the earliest sibling unblock instead of failing the delay cap", async () => { // Regression: with every sibling credential momentarily blocked (e.g. a // short post-401 or usage-probe block), a usage-limit 429 with a diff --git a/packages/coding-agent/test/auth-storage-rotation.test.ts b/packages/coding-agent/test/auth-storage-rotation.test.ts index c1278e9b3..2a48776de 100644 --- a/packages/coding-agent/test/auth-storage-rotation.test.ts +++ b/packages/coding-agent/test/auth-storage-rotation.test.ts @@ -95,4 +95,28 @@ describe("AuthStorage account rotation", () => { const exhaustedFallbackKey = await authStorage.getApiKey("openai-codex", sessionId); expect(exhaustedFallbackKey).toMatch(/^api-acct-/); }); + + test("usage-limit rotation can match the failed bearer when session stickiness is missing", async () => { + await authStorage.set("openai-codex", [ + { + type: "oauth", + access: "access-1", + refresh: "refresh-1", + expires: Date.now() + 60_000, + accountId: "acct-1", + }, + { + type: "oauth", + access: "access-2", + refresh: "refresh-2", + expires: Date.now() + 60_000, + accountId: "acct-2", + }, + ]); + + const sessionId = "missing-sticky-session"; + const result = await authStorage.markUsageLimitReached("openai-codex", sessionId, { apiKey: "access-1" }); + expect(result.switched).toBe(true); + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-2"); + }); }); From 0c6af8f408497c86ee32514ae3f6e1ca7257b1f3 Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Wed, 8 Jul 2026 00:39:25 +0800 Subject: [PATCH 66/91] fix(auth): rotate the failed bearer on usage limits --- packages/ai/src/auth-retry.ts | 7 +++-- packages/ai/src/auth-storage.ts | 8 +++--- packages/ai/src/stream.ts | 8 +++++- .../src/config/api-key-resolver.ts | 9 +++++-- .../test/auth-storage-rotation.test.ts | 26 +++++++++++++++++++ 5 files changed, 50 insertions(+), 8 deletions(-) diff --git a/packages/ai/src/auth-retry.ts b/packages/ai/src/auth-retry.ts index dc35d6a1c..87da67210 100644 --- a/packages/ai/src/auth-retry.ts +++ b/packages/ai/src/auth-retry.ts @@ -23,6 +23,8 @@ export interface ApiKeyResolveContext { lastChance: boolean; /** The auth error that triggered this re-resolution, or `undefined` on the initial resolve. */ error: unknown; + /** Bearer used by the failed attempt, when the caller can expose it. */ + previousKey?: string; /** Caller cancel signal, threaded into any credential refresh / rotation work. */ signal?: AbortSignal; } @@ -87,9 +89,10 @@ export async function resolveRetryKey( lastChance: boolean, error: unknown, signal?: AbortSignal, + previousKey?: string, ): Promise { try { - return (await resolver({ lastChance, error, signal })) || undefined; + return (await resolver({ lastChance, error, signal, previousKey })) || undefined; } catch { return undefined; } @@ -136,7 +139,7 @@ export async function withAuth( } for (let i = 0; i < AUTH_RETRY_STEPS.length; i++) { - const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal); + const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal, lastKey); if (nextKey === undefined || nextKey === lastKey) continue; lastKey = nextKey; try { diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 38c896e7e..958979e2f 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3059,8 +3059,8 @@ export class AuthStorage { sessionId: string | undefined, options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; apiKey?: string; signal?: AbortSignal }, ): Promise { - let sessionCredential = this.#getSessionCredential(provider, sessionId); - if (!sessionCredential && options?.apiKey) { + let sessionCredential: { type: AuthCredential["type"]; index: number } | undefined; + if (options?.apiKey) { const stored = this.#getStoredCredentials(provider); for (let index = 0; index < stored.length; index++) { const entry = stored[index]; @@ -3070,6 +3070,7 @@ export class AuthStorage { } } } + sessionCredential ??= this.#getSessionCredential(provider, sessionId); if (!sessionCredential) return { switched: false }; const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type); @@ -4391,7 +4392,7 @@ export class AuthStorage { async rotateSessionCredential( provider: string, sessionId: string | undefined, - options?: { error?: unknown; modelId?: string; signal?: AbortSignal }, + options?: { error?: unknown; modelId?: string; apiKey?: string; signal?: AbortSignal }, ): Promise { const sessionCredential = this.#getSessionCredential(provider, sessionId); if (!sessionCredential) return false; @@ -4403,6 +4404,7 @@ export class AuthStorage { return ( await this.markUsageLimitReached(provider, sessionId, { modelId: options?.modelId, + apiKey: options?.apiKey, signal: options?.signal, }) ).switched; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 286ac23a4..02d17ad2f 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1106,7 +1106,13 @@ export function streamSimple( // Caller aborted between attempts: don't mint a fresh token or fire // another doomed request — emit the captured failure instead. if (signal?.aborted) break; - const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal); + const nextKey = await resolveRetryKey( + apiKeyResolver, + AUTH_RETRY_STEPS[step]!, + failure.error, + signal, + lastKey, + ); if (nextKey === undefined || nextKey === lastKey) continue; lastKey = nextKey; const isLastStep = step === AUTH_RETRY_STEPS.length - 1; diff --git a/packages/coding-agent/src/config/api-key-resolver.ts b/packages/coding-agent/src/config/api-key-resolver.ts index f06bccc7e..d237302cb 100644 --- a/packages/coding-agent/src/config/api-key-resolver.ts +++ b/packages/coding-agent/src/config/api-key-resolver.ts @@ -49,7 +49,7 @@ export function createApiKeyResolver( options: ApiKeyResolverOptions = {}, ): ApiKeyResolver { const { sessionId, baseUrl, modelId } = options; - return async ({ lastChance, error, signal }) => { + return async ({ lastChance, error, signal, previousKey }) => { if (error === undefined) { return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId }); } @@ -59,7 +59,12 @@ export function createApiKeyResolver( // sibling exists we switch immediately; the precise no-sibling backoff // is owned by `markUsageLimitReached` (default + server usage-report // reset) and the outer whole-turn retry layer. - await registry.authStorage.rotateSessionCredential(provider, sessionId, { error, modelId, signal }); + await registry.authStorage.rotateSessionCredential(provider, sessionId, { + error, + modelId, + signal, + apiKey: previousKey, + }); return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId }); } return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId, forceRefresh: true, signal }); diff --git a/packages/coding-agent/test/auth-storage-rotation.test.ts b/packages/coding-agent/test/auth-storage-rotation.test.ts index 2a48776de..bf7e3e6c5 100644 --- a/packages/coding-agent/test/auth-storage-rotation.test.ts +++ b/packages/coding-agent/test/auth-storage-rotation.test.ts @@ -119,4 +119,30 @@ describe("AuthStorage account rotation", () => { expect(result.switched).toBe(true); expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-2"); }); + + test("usage-limit rotation trusts the failed bearer over stale session stickiness", async () => { + await authStorage.set("openai-codex", [ + { + type: "oauth", + access: "plus-access", + refresh: "plus-refresh", + expires: Date.now() + 60_000, + accountId: "plus-acct", + }, + { + type: "oauth", + access: "k12-access", + refresh: "k12-refresh", + expires: Date.now() + 60_000, + accountId: "k12-acct", + }, + ]); + + const sessionId = "stale-sticky-session"; + const stickyKey = await authStorage.getApiKey("openai-codex", sessionId); + const failedKey = stickyKey === "api-plus-acct" ? "k12-access" : "plus-access"; + const result = await authStorage.markUsageLimitReached("openai-codex", sessionId, { apiKey: failedKey }); + expect(result.switched).toBe(true); + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe(stickyKey); + }); }); From 169bc3683117afe6360012c0bbaa39afd2f84b53 Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Wed, 8 Jul 2026 00:49:05 +0800 Subject: [PATCH 67/91] fix(auth): share codex quota windows across plans --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/usage/openai-codex.ts | 2 -- packages/ai/test/openai-codex-usage.test.ts | 2 +- 3 files changed, 2 insertions(+), 3 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 06c456631..0a0f1dc7d 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -6,6 +6,7 @@ - Fixed access-token-only OAuth credentials attempting token refresh with an empty refresh token after expiry. - Fixed gateway usage-limit retries falling through to cross-provider model fallback before trying a sibling credential from the same provider. +- Fixed Codex usage-limit rotation treating Plus and K-12 accounts as separate quota groups for shared 5-hour/7-day windows. ## [16.3.11-zen.1] - 2026-07-07 diff --git a/packages/ai/src/usage/openai-codex.ts b/packages/ai/src/usage/openai-codex.ts index 716ddd5fb..355744208 100644 --- a/packages/ai/src/usage/openai-codex.ts +++ b/packages/ai/src/usage/openai-codex.ts @@ -285,8 +285,6 @@ function buildUsageLimit(args: { label: usageWindow.label, scope: { provider: "openai-codex", - accountId: args.accountId, - tier: args.planType, windowId: usageWindow.id, shared: true, }, diff --git a/packages/ai/test/openai-codex-usage.test.ts b/packages/ai/test/openai-codex-usage.test.ts index b4f385434..1027c8070 100644 --- a/packages/ai/test/openai-codex-usage.test.ts +++ b/packages/ai/test/openai-codex-usage.test.ts @@ -63,7 +63,7 @@ describe("openai-codex usage parser", () => { expect(report).not.toBeNull(); const main = report?.limits.filter(l => l.id === "openai-codex:primary" || l.id === "openai-codex:secondary"); expect(main?.map(l => l.id)).toEqual(["openai-codex:primary", "openai-codex:secondary"]); - expect(main?.[0].scope.tier).toBe("pro"); + expect(main?.[0].scope).toEqual({ provider: "openai-codex", windowId: "5h", shared: true }); expect(main?.[0].amount.usedFraction).toBeCloseTo(0.04, 5); }); From 4a48a0350690f0bafa57f464ee2400ba6863deb4 Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Wed, 8 Jul 2026 01:08:12 +0800 Subject: [PATCH 68/91] fix(auth): skip expired token-only sticky credentials --- packages/ai/src/auth-storage.ts | 9 ++++ .../test/auth-storage-codex-selection.test.ts | 50 ++++++++++++++++++- 2 files changed, 58 insertions(+), 1 deletion(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 958979e2f..3d0ef1262 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3398,12 +3398,21 @@ export class AuthStorage { const checkUsage = strategy !== undefined && (credentials.length > 1 || requiresProModel); const sessionCredential = this.#getSessionCredential(provider, sessionId); const sessionPreferredIndex = sessionCredential?.type === "oauth" ? sessionCredential.index : undefined; + const sessionPreferredCredential = + sessionPreferredIndex !== undefined + ? credentials.find(entry => entry.index === sessionPreferredIndex)?.credential + : undefined; + const sessionPreferredCanRefreshOrUse = + sessionPreferredCredential !== undefined && + (sessionPreferredCredential.refresh.trim().length > 0 || + Date.now() + OAUTH_REFRESH_SKEW_MS < sessionPreferredCredential.expires); // Skip ranking only when the session already has a working preferred credential — re-ranking // mid-session causes account switches that cold-start the server-side prompt cache. New sessions // (no preference) and sessions whose preferred is blocked still rank, so we pick the account // with the most headroom proactively and fall back intelligently when rate-limited. const sessionPreferredIsAvailable = sessionPreferredIndex !== undefined && + sessionPreferredCanRefreshOrUse && !this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope); const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel); const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6b803b400..3dd0a169f 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -98,7 +98,7 @@ function createCredential(accountId: string, email: string): OAuthCredentials { return { access: `access-${accountId}`, refresh: `refresh-${accountId}`, - expires: Date.now() + HOUR_MS, + expires: Date.now() + WEEK_MS, accountId, email, }; @@ -595,6 +595,54 @@ describe("AuthStorage codex oauth ranking", () => { // flaky on loaded CI runners, so maxConcurrent is the authoritative signal. expect(maxConcurrent).toBe(3); }); + + test("skips expired access-token-only sticky credential and selects fresh sibling", async () => { + if (!authStorage) throw new Error("test setup failed"); + const sessionId = "sticky-token-only-session"; + await authStorage.set("openai-codex", [{ type: "oauth", ...createCredential("acct-k12", "k12@example.com") }]); + usageByAccount.set( + "acct-k12", + createCodexUsageReport({ + accountId: "acct-k12", + primary: { usedFraction: 0.3, resetInMs: 20 * 60 * 1000 }, + secondary: { usedFraction: 0.2, resetInMs: 5 * 24 * 60 * 60 * 1000 }, + }), + ); + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-k12"); + usageByAccount.set( + "acct-k12", + createCodexUsageReport({ + accountId: "acct-k12", + primary: { usedFraction: 1, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 0.17, resetInMs: WEEK_MS }, + }), + ); + usageByAccount.set( + "acct-plus", + createCodexUsageReport({ + accountId: "acct-plus", + primary: { usedFraction: 0.2, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 0.74, resetInMs: WEEK_MS }, + }), + ); + + await authStorage.set("openai-codex", [ + { + type: "oauth", + access: "access-acct-k12", + refresh: "", + expires: Date.now() - 1_000, + accountId: "acct-k12", + email: "k12@example.com", + }, + { + type: "oauth", + ...createCredential("acct-plus", "plus@example.com"), + }, + ]); + + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-plus"); + }); }); // ───────────────────────────────────────────────────────────────────────────── From 0e259bb64010a8015ff34de5eb7b5a10184d6ecb Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Wed, 8 Jul 2026 01:39:41 +0800 Subject: [PATCH 69/91] fix(auth): rotate codex accounts before model fallback --- packages/ai/src/auth-retry.ts | 4 +- packages/ai/src/auth-storage.ts | 14 +++--- packages/ai/src/usage/openai-codex.ts | 3 ++ packages/ai/test/auth-retry.test.ts | 22 +++++++++ .../test/auth-storage-codex-selection.test.ts | 46 +++++++++++++++++++ 5 files changed, 82 insertions(+), 7 deletions(-) diff --git a/packages/ai/src/auth-retry.ts b/packages/ai/src/auth-retry.ts index 87da67210..22a8f8e55 100644 --- a/packages/ai/src/auth-retry.ts +++ b/packages/ai/src/auth-retry.ts @@ -1,6 +1,7 @@ import type { OAuthAccess } from "./auth-storage"; import * as AIError from "./error"; import { isAuthRetryableError } from "./error/auth-classify"; +import { isUsageLimit } from "./error/flags"; /** * Context passed to an {@link ApiKeyResolver} on each resolution attempt. @@ -92,7 +93,8 @@ export async function resolveRetryKey( previousKey?: string, ): Promise { try { - return (await resolver({ lastChance, error, signal, previousKey })) || undefined; + const rotateSibling = lastChance || (!lastChance && isUsageLimit(error)); + return (await resolver({ lastChance: rotateSibling, error, signal, previousKey })) || undefined; } catch { return undefined; } diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 3d0ef1262..67b698838 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -1390,12 +1390,14 @@ export class AuthStorage { const credentialId = this.#getStoredCredentials(provider)[credentialIndex]?.id; if (credentialId === undefined) return blockedUntil; - const persistedGlobalBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, ""); - if ( - persistedGlobalBlockedUntil !== undefined && - (blockedUntil === undefined || persistedGlobalBlockedUntil > blockedUntil) - ) { - blockedUntil = persistedGlobalBlockedUntil; + if (!blockScope || provider !== "openai-codex") { + const persistedGlobalBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, ""); + if ( + persistedGlobalBlockedUntil !== undefined && + (blockedUntil === undefined || persistedGlobalBlockedUntil > blockedUntil) + ) { + blockedUntil = persistedGlobalBlockedUntil; + } } if (blockScope) { const persistedScopedBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, blockScope); diff --git a/packages/ai/src/usage/openai-codex.ts b/packages/ai/src/usage/openai-codex.ts index 355744208..c7adebb4c 100644 --- a/packages/ai/src/usage/openai-codex.ts +++ b/packages/ai/src/usage/openai-codex.ts @@ -505,6 +505,9 @@ export const openaiCodexUsageProvider: UsageProvider = { const FIVE_HOUR_MS = 5 * 60 * 60 * 1000; export const codexRankingStrategy: CredentialRankingStrategy = { + blockScope() { + return "shared"; + }, findWindowLimits(report) { const findLimit = (key: "primary" | "secondary"): UsageLimit | undefined => { const direct = report.limits.find(l => l.id === `openai-codex:${key}`); diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts index 73b1f66d1..44a4a9f89 100644 --- a/packages/ai/test/auth-retry.test.ts +++ b/packages/ai/test/auth-retry.test.ts @@ -99,6 +99,28 @@ describe("withAuth", () => { ]); }); + it("switches accounts before refreshing the same account on usage limits", async () => { + const keys: string[] = []; + const contexts: ApiKeyResolveContext[] = []; + const result = await withAuth( + ctx => { + contexts.push(ctx); + return ctx.error === undefined ? "k0" : ctx.lastChance ? "k2" : "k1"; + }, + async key => { + keys.push(key); + if (key === "k2") return "success"; + throw usageLimitError(); + }, + ); + expect(result).toBe("success"); + expect(keys).toEqual(["k0", "k2"]); + expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ + { lastChance: false, hasError: false }, + { lastChance: true, hasError: true }, + ]); + }); + it("stops retrying when the resolver returns undefined", async () => { const keys: string[] = []; const original = authError(); diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 3dd0a169f..b6aa79847 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -643,6 +643,52 @@ describe("AuthStorage codex oauth ranking", () => { expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-plus"); }); + + test("ignores legacy global Codex blocks when a scoped quota window has fresh siblings", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-k12", "k12@example.com") }, + { type: "oauth", ...createCredential("acct-plus", "plus@example.com") }, + ]); + usageByAccount.set( + "acct-k12", + createCodexUsageReport({ + accountId: "acct-k12", + primary: { usedFraction: 1, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 1, resetInMs: WEEK_MS }, + }), + ); + usageByAccount.set( + "acct-plus", + createCodexUsageReport({ + accountId: "acct-plus", + primary: { usedFraction: 0.2, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 0.74, resetInMs: WEEK_MS }, + }), + ); + const plus = store + .listAuthCredentials("openai-codex") + .find(row => row.credential.type === "oauth" && row.credential.accountId === "acct-plus"); + if (!plus || !store.upsertCredentialBlock) throw new Error("missing plus credential row"); + store.upsertCredentialBlock({ + credentialId: plus.id, + providerKey: "openai-codex:oauth", + blockScope: "", + blockedUntilMs: Date.now() + WEEK_MS, + }); + const k12 = store + .listAuthCredentials("openai-codex") + .find(row => row.credential.type === "oauth" && row.credential.accountId === "acct-k12"); + if (!k12 || !store.upsertCredentialBlock) throw new Error("missing k12 credential row"); + store.upsertCredentialBlock({ + credentialId: k12.id, + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs: Date.now() + HOUR_MS, + }); + + expect(await authStorage.getApiKey("openai-codex", "session-with-legacy-global-block")).toBe("api-acct-plus"); + }); }); // ───────────────────────────────────────────────────────────────────────────── From e862c658972de72ef6ad6ddb3cf85158944ab967 Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Wed, 8 Jul 2026 01:59:04 +0800 Subject: [PATCH 70/91] test(auth): update usage-limit retry expectations --- packages/ai/test/stream-auth-retry.test.ts | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index 484d80f01..856fb08ae 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -405,7 +405,7 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["credential-A", "credential-B"]); expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); - expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([false, true]); + expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([true]); } }); @@ -501,10 +501,9 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "next-key"]); expect(retryContexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ - { lastChance: false, hasError: true }, { lastChance: true, hasError: true }, ]); - expect((retryContexts[1]?.error as Error).message).toContain("Resource exhausted"); + expect((retryContexts[0]?.error as Error).message).toContain("Resource exhausted"); }); it("surfaces the original error when the resolver declines every retry", async () => { From e787a90a7daf2bb7145b830c7ce335772d7fc056 Mon Sep 17 00:00:00 2001 From: qfrtt Date: Tue, 7 Jul 2026 15:03:55 -0400 Subject: [PATCH 71/91] Fix Edge browser launch --- packages/coding-agent/CHANGELOG.md | 4 +++ .../coding-agent/src/tools/browser/launch.ts | 35 ++++++++++++++++--- .../test/tools/browser-launch.test.ts | 35 +++++++++++++++++++ 3 files changed, 70 insertions(+), 4 deletions(-) create mode 100644 packages/coding-agent/test/tools/browser-launch.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..3291d31f0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed the browser tool failing to launch Microsoft Edge-only Windows installs with Puppeteer's empty `Code: 0` launch error by keeping Edge's required `--enable-automation` default while preserving Chrome/Chromium stealth launch defaults. + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/tools/browser/launch.ts b/packages/coding-agent/src/tools/browser/launch.ts index d9c9bb846..38344bff7 100644 --- a/packages/coding-agent/src/tools/browser/launch.ts +++ b/packages/coding-agent/src/tools/browser/launch.ts @@ -29,15 +29,20 @@ export const DEFAULT_VIEWPORT = { width: 1365, height: 768, deviceScaleFactor: 1 * connection dropped, etc.). */ export const BROWSER_PROTOCOL_TIMEOUT_MS = 60_000; +const ENABLE_AUTOMATION_FLAG = "--enable-automation"; // Automation-tell launch flags that puppeteer-core adds by default. We suppress // them via `ignoreDefaultArgs` (the supported escape hatch) to mirror xxxx's -// chromiumSwitches patch. `--enable-automation` is the loudest: it sets +// chromiumSwitches patch. `--enable-automation` is the loudest: it normally sets // navigator.webdriver=true and shows the "controlled by automated software" infobar. +// Edge is the launch-stability exception: it can exit before CDP opens when this +// default flag is stripped, so Edge keeps Puppeteer's flag while our explicit +// `--disable-blink-features=AutomationControlled` launch arg still handles +// navigator.webdriver. // `ignoreDefaultArgs` does exact-string matching, so each entry must be a flag that // puppeteer emits verbatim. The default `--disable-features=...` string can't be // matched this way; it is neutralized in the puppeteer-core patch (ChromeLauncher). const STEALTH_IGNORE_DEFAULT_ARGS = [ - "--enable-automation", + ENABLE_AUTOMATION_FLAG, "--disable-extensions", "--disable-default-apps", "--disable-component-extensions-with-background-pages", @@ -47,6 +52,23 @@ const STEALTH_IGNORE_DEFAULT_ARGS = [ "--disable-ipc-flooding-protection", "--metrics-recording-only", ]; + +function isMicrosoftEdgeExecutable(executablePath: string | undefined): boolean { + if (!executablePath) return false; + const normalizedPath = executablePath.replaceAll("\\", "/").toLowerCase(); + const executableName = normalizedPath.slice(normalizedPath.lastIndexOf("/") + 1); + return ( + executableName === "msedge.exe" || + executableName === "microsoft edge" || + executableName.startsWith("microsoft-edge") + ); +} + +function stealthIgnoreDefaultArgs(executablePath: string | undefined): string[] { + if (!isMicrosoftEdgeExecutable(executablePath)) return [...STEALTH_IGNORE_DEFAULT_ARGS]; + return STEALTH_IGNORE_DEFAULT_ARGS.filter(arg => arg !== ENABLE_AUTOMATION_FLAG); +} + const STEALTH_ACCEPT_LANGUAGE = "en-US,en"; const USER_AGENT_TARGET_TIMEOUT_MS = 5_000; @@ -282,12 +304,13 @@ export async function launchHeadlessBrowser(opts: LaunchHeadlessOptions): Promis if (ignoreCert === "true" || ignoreCert === "1" || ignoreCert === "yes" || ignoreCert === "on") { launchArgs.push("--ignore-certificate-errors"); } + const executablePath = await ensureChromiumExecutable(); return await puppeteer.launch({ headless: opts.headless, defaultViewport: opts.headless ? initialViewport : null, - executablePath: await ensureChromiumExecutable(), + executablePath, args: launchArgs, - ignoreDefaultArgs: [...STEALTH_IGNORE_DEFAULT_ARGS], + ignoreDefaultArgs: stealthIgnoreDefaultArgs(executablePath), protocolTimeout: BROWSER_PROTOCOL_TIMEOUT_MS, }); } @@ -737,6 +760,10 @@ export async function applyStealthPatches( await injectStealthScripts(page); } +export function stealthIgnoreDefaultArgsForTest(executablePath: string | undefined): string[] { + return stealthIgnoreDefaultArgs(executablePath); +} + export function targetSupportsUserAgentOverrideForTest(target: Target): boolean { return targetSupportsUserAgentOverride(target); } diff --git a/packages/coding-agent/test/tools/browser-launch.test.ts b/packages/coding-agent/test/tools/browser-launch.test.ts new file mode 100644 index 000000000..1d677507f --- /dev/null +++ b/packages/coding-agent/test/tools/browser-launch.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it } from "bun:test"; +import { stealthIgnoreDefaultArgsForTest } from "@oh-my-pi/pi-coding-agent/tools/browser/launch"; + +const AUTOMATION_FLAG = "--enable-automation"; + +const EDGE_EXECUTABLE_PATHS = [ + "C:\\Program Files\\Microsoft\\Edge\\Application\\msedge.exe", + "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge", + "/usr/bin/microsoft-edge-stable", +] as const; + +const CHROME_EXECUTABLE_PATHS = [ + "C:\\Program Files\\Google\\Chrome\\Application\\chrome.exe", + "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", + "/usr/bin/chromium", +] as const; + +describe("browser launch stealth defaults", () => { + it("keeps Puppeteer's automation default for Microsoft Edge executables", () => { + for (const executablePath of EDGE_EXECUTABLE_PATHS) { + const ignoreDefaultArgs = stealthIgnoreDefaultArgsForTest(executablePath); + + expect(ignoreDefaultArgs).not.toContain(AUTOMATION_FLAG); + expect(ignoreDefaultArgs).toContain("--disable-extensions"); + } + }); + + it("continues filtering Puppeteer's automation default for Chrome and Chromium executables", () => { + for (const executablePath of CHROME_EXECUTABLE_PATHS) { + const ignoreDefaultArgs = stealthIgnoreDefaultArgsForTest(executablePath); + + expect(ignoreDefaultArgs).toContain(AUTOMATION_FLAG); + } + }); +}); From 9cb051a4b17249a9fe3e8801b36a97fabef8b84b Mon Sep 17 00:00:00 2001 From: ben Date: Wed, 8 Jul 2026 18:32:05 +0800 Subject: [PATCH 72/91] Fix JS worker cwd conflict handling --- .../coding-agent/src/eval/js/worker-core.ts | 18 +-- .../test/eval/worker-core.test.ts | 114 ++++++++++++++++++ 2 files changed, 123 insertions(+), 9 deletions(-) create mode 100644 packages/coding-agent/test/eval/worker-core.test.ts diff --git a/packages/coding-agent/src/eval/js/worker-core.ts b/packages/coding-agent/src/eval/js/worker-core.ts index 4b7933d3c..74511e696 100644 --- a/packages/coding-agent/src/eval/js/worker-core.ts +++ b/packages/coding-agent/src/eval/js/worker-core.ts @@ -77,16 +77,16 @@ export class WorkerCore { } async #runOne(runId: string, code: string, filename: string, snapshot: SessionSnapshot): Promise { - const runtime = this.#ensureRuntime(snapshot); - runtime.setCwd(snapshot.cwd); - const active: ActiveRun = { runId, pendingTools: new Map() }; - this.#runs.set(runId, active); - const hooks: RuntimeHooks = { - onText: chunk => this.#transport.send({ type: "text", runId, chunk }), - onDisplay: output => this.#transport.send({ type: "display", runId, output }), - callTool: (name, args) => this.#callTool(active, name, args), - }; try { + const runtime = this.#ensureRuntime(snapshot); + runtime.setCwd(snapshot.cwd); + const active: ActiveRun = { runId, pendingTools: new Map() }; + this.#runs.set(runId, active); + const hooks: RuntimeHooks = { + onText: chunk => this.#transport.send({ type: "text", runId, chunk }), + onDisplay: output => this.#transport.send({ type: "display", runId, output }), + callTool: (name, args) => this.#callTool(active, name, args), + }; const value = await runtime.run(code, filename, hooks, { runId, cwd: snapshot.cwd }); runtime.displayValue(value, hooks); this.#transport.send({ type: "result", runId, ok: true }); diff --git a/packages/coding-agent/test/eval/worker-core.test.ts b/packages/coding-agent/test/eval/worker-core.test.ts new file mode 100644 index 000000000..ab1d71846 --- /dev/null +++ b/packages/coding-agent/test/eval/worker-core.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, it } from "bun:test"; +import { WorkerCore } from "@oh-my-pi/pi-coding-agent/eval/js/worker-core"; +import type { + SessionSnapshot, + Transport, + WorkerInbound, + WorkerOutbound, +} from "@oh-my-pi/pi-coding-agent/eval/js/worker-protocol"; + +interface WorkerHarness { + send(message: WorkerInbound): void; + onMessage(handler: (message: WorkerOutbound) => void): () => void; +} + +function createWorkerHarness(): WorkerHarness { + const hostListeners = new Set<(message: WorkerOutbound) => void>(); + const workerListeners = new Set<(message: WorkerInbound) => void>(); + const transport: Transport = { + send: message => { + queueMicrotask(() => { + for (const listener of hostListeners) listener(message); + }); + }, + onMessage: handler => { + workerListeners.add(handler); + return () => workerListeners.delete(handler); + }, + close: () => {}, + }; + new WorkerCore(transport); + return { + send(message) { + queueMicrotask(() => { + for (const listener of workerListeners) listener(message); + }); + }, + onMessage(handler) { + hostListeners.add(handler); + return () => hostListeners.delete(handler); + }, + }; +} + +function waitForMessage( + harness: WorkerHarness, + predicate: (message: WorkerOutbound) => boolean, +): Promise { + const { promise, resolve } = Promise.withResolvers(); + let unsubscribe = (): void => {}; + unsubscribe = harness.onMessage(message => { + if (!predicate(message)) return; + unsubscribe(); + resolve(message); + }); + return promise; +} + +async function initializeWorker(harness: WorkerHarness, snapshot: SessionSnapshot): Promise { + const ready = waitForMessage(harness, message => message.type === "ready"); + harness.send({ type: "init", snapshot }); + expect((await ready).type).toBe("ready"); +} + +describe("WorkerCore", () => { + it("reports same-realm cwd conflicts through the worker protocol", async () => { + const first = createWorkerHarness(); + const second = createWorkerHarness(); + const cwd = process.cwd(); + await initializeWorker(first, { cwd, sessionId: "same-realm-first", localRoots: {} }); + await initializeWorker(second, { cwd, sessionId: "same-realm-second", localRoots: {} }); + + const gate = Promise.withResolvers(); + const entered = Promise.withResolvers(); + (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }).__omp_worker_core_gate = { + entered: () => entered.resolve(), + wait: gate.promise, + }; + try { + first.send({ + type: "run", + runId: "hold-first-runtime", + code: "globalThis.__omp_worker_core_gate.entered(); await globalThis.__omp_worker_core_gate.wait;", + filename: "[same-realm-first].js", + snapshot: { cwd, sessionId: "same-realm-first", localRoots: {} }, + }); + await entered.promise; + + const result = waitForMessage( + second, + message => message.type === "result" && message.runId === "overlap-second-runtime", + ); + second.send({ + type: "run", + runId: "overlap-second-runtime", + code: "1 + 1;", + filename: "[same-realm-second].js", + snapshot: { cwd, sessionId: "same-realm-second", localRoots: {} }, + }); + + expect(await result).toMatchObject({ + type: "result", + runId: "overlap-second-runtime", + ok: false, + error: { message: "Cannot set cwd while another same-realm JS runtime is running" }, + }); + } finally { + gate.resolve(); + delete (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }) + .__omp_worker_core_gate; + first.send({ type: "close" }); + second.send({ type: "close" }); + } + }); +}); From 38486e56db1604422b777a375856c3b556e6ecfb Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 14:51:25 +0200 Subject: [PATCH 73/91] feat(coding-agent): handled unawaited promise rejections in eval cells - Introduced a rejection interception mechanism to capture unhandled promise rejections from eval cell code. - Attributed floating rejections to specific runs to fail the owning cell instead of crashing the process or worker. - Downgraded rejections occurring after a cell finished to warn logs to prevent silent failures. --- packages/coding-agent/CHANGELOG.md | 6 +- .../coding-agent/src/eval/js/worker-core.ts | 165 +++++++++++++++++- packages/utils/CHANGELOG.md | 4 + packages/utils/src/postmortem.ts | 27 +++ 4 files changed, 197 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bde1313a3..cdd0f2940 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,12 +5,16 @@ ### Changed - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. + ### Fixed -- Improved advisor robustness by blocking exhausted accounts during consecutive turn failures +- Improved handling of unawaited promises in JS eval cells to prevent process crashes +- Added warning logs for unhandled rejections originating from finished eval cells +- Improved advisor robustness by blocking exhausted accounts during consecutive turn failures - Fixed advisor turns hammering the same usage-limited account: a failed advisor turn now marks the exhausted credential blocked (with the provider's retry hint and usage-report reset time), so the next retry rotates to a sibling instead of re-picking the blocked account every few seconds. Previously the in-stream auth retry rotated within a request but never blocked the last failing credential, and the advisor loop — unlike the primary retry pipeline — never called `markUsageLimitReached`. - Added the account key to the `codex-auto-reset: skipped` debug log so skip reasons (e.g. `weekly-not-exhausted`) can be attributed to the evaluated account. +- Fixed unawaited promise rejections in JS eval cells crashing the session: a floating rejection now fails the owning cell run (`Unhandled rejection (missing await?): …`) instead of escaping to the global `unhandledRejection` handler, which printed `[Unhandled Rejection]` and killed the process (inline fallback) or tore down the eval worker (dedicated worker). Rejections surfacing after a cell settled are downgraded to a warn log attributed to the finished cell. ## [16.3.11] - 2026-07-06 diff --git a/packages/coding-agent/src/eval/js/worker-core.ts b/packages/coding-agent/src/eval/js/worker-core.ts index 4b7933d3c..437fbd6d3 100644 --- a/packages/coding-agent/src/eval/js/worker-core.ts +++ b/packages/coding-agent/src/eval/js/worker-core.ts @@ -1,6 +1,15 @@ +import { isMainThread } from "node:worker_threads"; +import { postmortem } from "@oh-my-pi/pi-utils"; import { ToolError } from "../../tools/tool-errors"; import { JsRuntime, type RuntimeHooks } from "./shared/runtime"; -import type { RunErrorPayload, SessionSnapshot, ToolReply, Transport, WorkerInbound } from "./worker-protocol"; +import type { + RunErrorPayload, + SessionSnapshot, + ToolReply, + Transport, + WorkerInbound, + WorkerOutbound, +} from "./worker-protocol"; interface PendingTool { runId: string; @@ -10,9 +19,17 @@ interface PendingTool { interface ActiveRun { runId: string; + filename: string; pendingTools: Map; + /** Rejections floated by this run's cell code, captured before its result was sent. */ + floatingRejections: unknown[]; } +type RunResult = Extract; + +/** Finished-cell filenames retained for attributing rejections that surface after the run settled. */ +const RECENT_CELL_FILES_MAX = 256; + function errorPayload(error: unknown): RunErrorPayload { if (error instanceof Error) { return { @@ -34,15 +51,134 @@ function errorFromPayload(payload: RunErrorPayload): Error { return error; } +/** + * Fold rejections floated by cell code into the run result: an otherwise + * successful run fails with the first floating rejection (an unawaited promise + * failing is a cell failure, not a success with noise); the rest surface as + * output text so nothing is silently dropped. + */ +function foldFloatingRejections(active: ActiveRun, result: RunResult, hooks: RuntimeHooks): RunResult { + const rejections = active.floatingRejections; + if (rejections.length === 0) return result; + let folded = result; + let reported = rejections; + if (result.ok) { + const error = errorPayload(rejections[0]); + error.message = `Unhandled rejection (missing await?): ${error.message}`; + folded = { type: "result", runId: active.runId, ok: false, error }; + reported = rejections.slice(1); + } + for (const reason of reported) { + const payload = errorPayload(reason); + hooks.onText(`[unhandled rejection] ${payload.name ?? "Error"}: ${payload.message}\n`); + } + return folded; +} + export class WorkerCore { #transport: Transport; #runtime: JsRuntime | null = null; #runs = new Map(); + #recentCellFiles = new Set(); #unsubscribe: () => void; + #uninstallRejectionGuard: () => void; constructor(transport: Transport) { this.#transport = transport; this.#unsubscribe = transport.onMessage(msg => this.#handle(msg)); + this.#uninstallRejectionGuard = this.#installRejectionGuard(); + } + + /** + * Capture unhandled rejections floated by eval-cell code (unawaited async + * calls) so they fail the owning run instead of tearing down the worker or — + * via the global postmortem handler — the whole session. On the main thread + * (inline fallback) only cell-attributable rejections are consumed; in the + * dedicated worker realm a rejection during a live run is cell activity even + * without a usable stack, while anything else keeps its default fatality. + */ + #installRejectionGuard(): () => void { + if (isMainThread) { + return postmortem.interceptUnhandledRejections(reason => this.#consumeRejection(reason)); + } + const onRejection = (reason: unknown): void => { + if (this.#consumeRejection(reason)) return; + // Not cell-attributable: restore default fatality. Rethrowing from a + // timer surfaces it as an uncaught exception, which reaches the host + // as a worker `error` event exactly like an unhandled rejection did + // before this listener existed. + setTimeout(() => { + throw reason; + }, 0); + }; + process.on("unhandledRejection", onRejection); + return () => { + process.off("unhandledRejection", onRejection); + }; + } + + /** + * Attribute an unhandled rejection to eval-cell code. Live runs are stashed + * on the run (folded into its result after the settle drain); finished cells + * downgrade to a host-side warn log. Returns false when the rejection is not + * cell activity and must keep the default fatal path. + */ + #consumeRejection(reason: unknown): boolean { + const stack = reason instanceof Error && typeof reason.stack === "string" ? reason.stack : undefined; + if (stack) { + // The stack can name several cells (helper defined by an earlier cell, + // called from the live one); the outermost matching frame is the caller + // that owns the floating promise. + let owner: ActiveRun | undefined; + let ownerIndex = -1; + for (const run of this.#runs.values()) { + const index = stack.lastIndexOf(run.filename); + if (index > ownerIndex) { + ownerIndex = index; + owner = run; + } + } + if (owner) { + owner.floatingRejections.push(reason); + return true; + } + let recent: string | undefined; + let recentIndex = -1; + for (const filename of this.#recentCellFiles) { + const index = stack.lastIndexOf(filename); + if (index > recentIndex) { + recentIndex = index; + recent = filename; + } + } + if (recent) { + this.#transport.send({ + type: "log", + level: "warn", + msg: "Unhandled rejection from a finished eval cell (missing await?)", + meta: { filename: recent, error: errorPayload(reason) }, + }); + return true; + } + } + if (!isMainThread && this.#runs.size > 0) { + // Dedicated eval worker: during a live run, a rejection without a cell + // frame (e.g. `Promise.reject("msg")` or a library-created reason) is + // still cell activity — nothing else runs user code in this realm. + if (this.#runs.size === 1) { + const only = this.#runs.values().next().value; + only?.floatingRejections.push(reason); + return true; + } + this.#transport.send({ + type: "log", + level: "warn", + msg: "Unhandled rejection during concurrent eval runs; cannot attribute to a cell", + meta: { error: errorPayload(reason) }, + }); + return true; + } + return false; } #handle(msg: WorkerInbound): void { @@ -79,21 +215,40 @@ export class WorkerCore { async #runOne(runId: string, code: string, filename: string, snapshot: SessionSnapshot): Promise { const runtime = this.#ensureRuntime(snapshot); runtime.setCwd(snapshot.cwd); - const active: ActiveRun = { runId, pendingTools: new Map() }; + const active: ActiveRun = { runId, filename, pendingTools: new Map(), floatingRejections: [] }; this.#runs.set(runId, active); const hooks: RuntimeHooks = { onText: chunk => this.#transport.send({ type: "text", runId, chunk }), onDisplay: output => this.#transport.send({ type: "display", runId, output }), callTool: (name, args) => this.#callTool(active, name, args), }; + let result: RunResult; try { const value = await runtime.run(code, filename, hooks, { runId, cwd: snapshot.cwd }); runtime.displayValue(value, hooks); - this.#transport.send({ type: "result", runId, ok: true }); + result = { type: "result", runId, ok: true }; } catch (error) { - this.#transport.send({ type: "result", runId, ok: false, error: errorPayload(error) }); + result = { type: "result", runId, ok: false, error: errorPayload(error) }; + } + try { + // One event-loop turn so rejections the cell already floated surface + // while this run can still own them (rejection callbacks run before + // timers fire). + await Bun.sleep(0); + result = foldFloatingRejections(active, result, hooks); } finally { this.#runs.delete(runId); + this.#rememberCellFile(filename); + this.#transport.send(result); + } + } + + #rememberCellFile(filename: string): void { + this.#recentCellFiles.delete(filename); + this.#recentCellFiles.add(filename); + if (this.#recentCellFiles.size > RECENT_CELL_FILES_MAX) { + const oldest = this.#recentCellFiles.values().next().value; + if (oldest !== undefined) this.#recentCellFiles.delete(oldest); } } @@ -127,6 +282,7 @@ export class WorkerCore { this.#runtime?.dispose?.(); this.#runtime = null; this.#transport.send({ type: "closed" }); + this.#uninstallRejectionGuard(); this.#unsubscribe(); this.#transport.close(); } @@ -141,6 +297,7 @@ export class WorkerCore { this.#runs.clear(); this.#runtime?.dispose?.(); this.#runtime = null; + this.#uninstallRejectionGuard(); this.#unsubscribe(); try { this.#transport.close(); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 80ebafb3a..58fac53fc 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `postmortem.interceptUnhandledRejections()` to register interceptors consulted before an unhandled rejection tears the process down; a consuming interceptor (e.g. the JS eval runtime claiming rejections floated by user cell code) keeps the process alive and owns reporting. + ## [16.3.10] - 2026-07-06 ### Added diff --git a/packages/utils/src/postmortem.ts b/packages/utils/src/postmortem.ts index 01d208e9e..42c37b42e 100644 --- a/packages/utils/src/postmortem.ts +++ b/packages/utils/src/postmortem.ts @@ -121,6 +121,24 @@ export function isExpectedCleanupError(reason: unknown): boolean { return false; } +/** + * Interceptors consulted by the global `unhandledRejection` handler before the + * fatal path. See {@link interceptUnhandledRejections}. + */ +const rejectionInterceptors = new Set<(reason: unknown) => boolean>(); + +/** + * Register an interceptor consulted before an unhandled rejection tears the + * process down. Return `true` to consume the rejection — the interceptor owns + * reporting and the process continues. Used by embedded script runtimes (JS + * eval cells) whose user code can float rejections the host must not die for. + * Returns an unregister function. + */ +export function interceptUnhandledRejections(interceptor: (reason: unknown) => boolean): () => void { + rejectionInterceptors.add(interceptor); + return () => rejectionInterceptors.delete(interceptor); +} + function formatFatalError(label: string, err: Error): string { const name = err.name || "Error"; const message = err.message || "(no message)"; @@ -172,6 +190,15 @@ if (isMainThread) { logger.warn("Ignoring expected cleanup rejection", { err }); return; } + for (const interceptor of rejectionInterceptors) { + try { + if (interceptor(reason)) return; + } catch (interceptorErr) { + logger.warn("Unhandled-rejection interceptor threw; continuing with fatal path", { + err: interceptorErr, + }); + } + } process.stderr.write(formatFatalError("Unhandled Rejection", err)); logger.error("Unhandled rejection", { err }); await runCleanup(Reason.UNHANDLED_REJECTION); From 5f948878bb13e13b1f38d29ebf0bda07e248520f Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:03:06 +0200 Subject: [PATCH 74/91] fix(ai): guard codex stale-code lookup --- packages/ai/src/providers/openai-codex-responses.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 7b06efee4..5030cc67c 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -1242,7 +1242,7 @@ const CODEX_STALE_PREVIOUS_RESPONSE_CODES: Record = { function isCodexStalePreviousResponseError(error: unknown): boolean { if (!(error instanceof Error)) return false; - if ("code" in error && typeof error.code === "string" && CODEX_STALE_PREVIOUS_RESPONSE_CODES[error.code]) { + if ("code" in error && typeof error.code === "string" && Object.hasOwn(CODEX_STALE_PREVIOUS_RESPONSE_CODES, error.code)) { return true; } // Message-based fallback for providers/proxies that report the condition From d47879febedec6c7818d9f110dff67632eac6156 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 13:58:23 +0200 Subject: [PATCH 75/91] chore: drop unrelated changelog entries --- packages/ai/CHANGELOG.md | 10 ---------- 1 file changed, 10 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f80cf6969..d20b9ce7e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -9,16 +9,6 @@ - Fixed gateway usage-limit retries falling through to cross-provider model fallback before trying a sibling credential from the same provider. - Fixed Codex usage-limit rotation treating Plus and K-12 accounts as separate quota groups for shared 5-hour/7-day windows. -## [16.3.11-zen.1] - 2026-07-07 - -### Added - -- Added `anysearch` registry provider definition and interactive credentials login support. - -### Fixed - -- Fixed plain-text 5xx provider status messages (including relay `520` HTML pages) being left as non-retryable numeric statuses instead of transient retryable errors. -- Fixed relay `insufficient_user_*` quota errors being classified as generic 403 auth failures instead of retryable usage-limit errors. ## [16.3.11] - 2026-07-06 From 8471244655465da99c02d61f68050efde220f0be Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 14:55:07 +0200 Subject: [PATCH 76/91] fix(auth): honor login API keys in peek --- packages/ai/src/auth-storage.ts | 14 ++++++++++++-- .../ai/test/auth-storage-api-key-login.test.ts | 1 + 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 940d4516a..3076276fd 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3926,8 +3926,8 @@ export class AuthStorage { return configKey; } - // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored - // static api_key (which may be a stale broker-migrated copy) as a last resort. + // Precedence: a deliberate OAuth/login credential wins, then an explicit env var, + // then a stored static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthSelection = this.#selectCredentialByType(provider, "oauth"); if (oauthSelection) { const expiresAt = oauthSelection.credential.expires; @@ -3943,6 +3943,16 @@ export class AuthStorage { } } + const loginApiKeySelection = this.#selectCredentialByType( + provider, + "api_key", + undefined, + credential => credential.type === "api_key" && credential.source === "login", + ); + if (loginApiKeySelection) { + return this.#configValueResolver(loginApiKeySelection.credential.key); + } + const envKey = getEnvApiKey(provider); if (envKey) return envKey; diff --git a/packages/ai/test/auth-storage-api-key-login.test.ts b/packages/ai/test/auth-storage-api-key-login.test.ts index 5d4fe3c39..1cdb9f8fa 100644 --- a/packages/ai/test/auth-storage-api-key-login.test.ts +++ b/packages/ai/test/auth-storage-api-key-login.test.ts @@ -201,5 +201,6 @@ describe("AuthStorage api-key login upsert", () => { }); expect(await authStorage.getApiKey("opencode-go", "session-opencode-go-login")).toBe("new-opencode-key"); + expect(await authStorage.peekApiKey("opencode-go")).toBe("new-opencode-key"); }); }); From 6aa8aefb151e8e37162d23d0527d26303247db82 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 14:54:42 +0200 Subject: [PATCH 77/91] fix-catalog-litellm-cache-version --- packages/catalog/src/provider-models/openai-compat.ts | 11 ++++++----- packages/catalog/test/litellm-provider.test.ts | 4 ++-- 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 6be4c1f24..a33fe6cf9 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3339,11 +3339,12 @@ export function litellmModelManagerOptions( const baseUrl = config?.baseUrl ?? Bun.env.LITELLM_BASE_URL ?? "http://localhost:4000/v1"; return { providerId: "litellm", - // rich-v3 invalidates rows cached before reseller usage-suffix stripping - // and placeholder-only `all-team-models` filtering; bump the version whenever - // the mappers below change, or warm authoritative caches keep serving - // pre-change rows for the full TTL. - cacheProviderId: `litellm:rich-v3:${Bun.hash(baseUrl).toString(36)}`, + // rich-v4 invalidates rows cached before LiteLLM ids gained bundled + // reference fallback. Earlier versions also handled reseller usage-suffix + // stripping and placeholder-only `all-team-models` filtering; bump the + // version whenever the mappers below change, or warm authoritative caches + // keep serving pre-change rows for the full TTL. + cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy and is never bundled in models.json (that // would leak the machine's localhost catalog). Prefer the proxy's richer // management metadata, then enrich ids against models.dev with the bundled diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index dd9e042c6..02e7f8664 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -125,7 +125,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v3:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, + `litellm:rich-v4:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -148,7 +148,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v3:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, + `litellm:rich-v4:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); From 5cd4b68d63e3a3c307985accdf933100ec480145 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:10:07 +0200 Subject: [PATCH 78/91] fix(prompting): gate workflow notice on task tool --- .../coding-agent/src/session/agent-session.ts | 2 +- .../test/agent-session-magic-keywords.test.ts | 30 +++++++++++++++++-- 2 files changed, 28 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 051fbea7f..5fc4697f9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -7370,7 +7370,7 @@ export class AgentSession { timestamp, }); } - if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text)) { + if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text) && this.getActiveToolNames().includes("task")) { keywordNotices.push({ role: "custom", customType: "workflow-notice", diff --git a/packages/coding-agent/test/agent-session-magic-keywords.test.ts b/packages/coding-agent/test/agent-session-magic-keywords.test.ts index 4ac59444d..d675005ed 100644 --- a/packages/coding-agent/test/agent-session-magic-keywords.test.ts +++ b/packages/coding-agent/test/agent-session-magic-keywords.test.ts @@ -2,7 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; -import { Agent } from "@oh-my-pi/pi-agent-core"; +import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; import { Effort } from "@oh-my-pi/pi-ai"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as autoThinkingClassifier from "@oh-my-pi/pi-coding-agent/auto-thinking/classifier"; @@ -12,9 +12,21 @@ import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking"; +import { type } from "arktype"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; -async function createMagicKeywordSession(root: string): Promise<{ +const mockTaskTool: AgentTool = { + name: "task", + label: "Task", + description: "Mock task tool", + parameters: type({}), + execute: async () => ({ content: [{ type: "text" as const, text: "ok" }] }), +}; + +async function createMagicKeywordSession( + root: string, + tools: AgentTool[] = [mockTaskTool], +): Promise<{ session: AgentSession; settings: Settings; authStorage: AuthStorage; @@ -25,7 +37,7 @@ async function createMagicKeywordSession(root: string): Promise<{ initialState: { model, systemPrompt: ["Test"], - tools: [], + tools, messages: [], thinkingLevel: Effort.High, }, @@ -119,6 +131,18 @@ describe("AgentSession magic keyword settings", () => { expect(notice).not.toContain("Call `task` once per independent fan-out batch"); }); + it("skips workflowz notice when the task tool is inactive", async () => { + const created = await createMagicKeywordSession(root, []); + session = created.session; + authStorage = created.authStorage; + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + + await session.prompt("please workflowz this"); + + const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ customType?: string }>; + expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]); + }); + it("does not use a disabled ultrathink keyword to force auto thinking", async () => { const created = await createMagicKeywordSession(root); session = created.session; From 41f2074ec4c3b0102b29bff8958c27f96997ffbf Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 14:55:53 +0200 Subject: [PATCH 79/91] fix(coding-agent): align python prompt mode detection --- .../src/modes/controllers/input-controller.ts | 2 +- .../test/input-controller-keybindings.test.ts | 17 +++++++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index 9fd238744..aa70e59ca 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -552,7 +552,7 @@ export class InputController { const wasPythonMode = this.ctx.isPythonMode; const trimmed = text.trimStart(); this.ctx.isBashMode = trimmed.startsWith("!"); - this.ctx.isPythonMode = pythonCommandPrefixLength(trimmed) > 0; + this.ctx.isPythonMode = parsePythonCommandInput(trimmed) !== undefined; if (wasBashMode !== this.ctx.isBashMode || wasPythonMode !== this.ctx.isPythonMode) { this.ctx.updateEditorBorderColor(); } diff --git a/packages/coding-agent/test/input-controller-keybindings.test.ts b/packages/coding-agent/test/input-controller-keybindings.test.ts index 772dbf9cd..e0f1e9382 100644 --- a/packages/coding-agent/test/input-controller-keybindings.test.ts +++ b/packages/coding-agent/test/input-controller-keybindings.test.ts @@ -251,6 +251,23 @@ describe("InputController keybinding setup", () => { expect(spies.resetDisplay).toHaveBeenCalledTimes(1); }); + it("does not mark pasted shell prompts as Python mode while editing", async () => { + const { InputController, ctx, editor } = await createContext(); + const controller = new InputController(ctx); + + controller.setupKeyHandlers(); + + editor.onChange?.("$ cd ~/project && sudo ./build-and-push.sh o5.7 2>&1 | tail -4"); + + expect(ctx.isPythonMode).toBe(false); + expect(ctx.updateEditorBorderColor).not.toHaveBeenCalled(); + + editor.onChange?.("$ print(1)"); + + expect(ctx.isPythonMode).toBe(true); + expect(ctx.updateEditorBorderColor).toHaveBeenCalledTimes(1); + }); + it("registers retry as an editor action and retries the failed turn", async () => { const { InputController, ctx, editor, spies } = await createContext(); const controller = new InputController(ctx); From 7e4360d3173327e511bcc638ac6441eb1a1b3fb0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:09:20 +0200 Subject: [PATCH 80/91] fix(discovery): preserve marketplace-root skill selection --- .../src/discovery/claude-plugins.ts | 24 +++++++- .../test/discovery/claude-plugins.test.ts | 55 +++++++++++++++++++ 2 files changed, 78 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/discovery/claude-plugins.ts b/packages/coding-agent/src/discovery/claude-plugins.ts index 7cfd8ad83..3bb09292d 100644 --- a/packages/coding-agent/src/discovery/claude-plugins.ts +++ b/packages/coding-agent/src/discovery/claude-plugins.ts @@ -55,6 +55,27 @@ async function readPluginManifest(root: ClaudePluginRoot): Promise { + return value !== null && typeof value === "object" && !Array.isArray(value); +} + +async function skillsManifestReplacesFallback(root: ClaudePluginRoot): Promise { + const raw = await readFile(path.join(root.path, "marketplace.json")); + if (raw === null) return false; + + try { + const parsed: unknown = JSON.parse(raw); + if (!isRecord(parsed)) return false; + const plugins = parsed.plugins; + return ( + Array.isArray(plugins) && + plugins.some(entry => isRecord(entry) && entry.name === root.plugin && entry.source === "./") + ); + } catch { + return false; + } +} + function isWithinPluginRoot(rootPath: string, targetPath: string): boolean { const relative = path.relative(rootPath, targetPath); return relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative)); @@ -157,11 +178,12 @@ async function loadSkills(ctx: LoadContext): Promise> { warnings.push(...rootWarnings); const results = await Promise.all( roots.map(async root => { + const includeFallback = !(await skillsManifestReplacesFallback(root)); const { dirs: skillsDirs, warnings: resolveWarnings } = await resolvePluginDir( root, ["skills"], "skills", - true, + includeFallback, ); const scanResults = await Promise.all( skillsDirs.map(dir => diff --git a/packages/coding-agent/test/discovery/claude-plugins.test.ts b/packages/coding-agent/test/discovery/claude-plugins.test.ts index b7afdd239..371c2e039 100644 --- a/packages/coding-agent/test/discovery/claude-plugins.test.ts +++ b/packages/coding-agent/test/discovery/claude-plugins.test.ts @@ -859,6 +859,61 @@ describe("listClaudePluginRoots", () => { expect(result.all.find(s => s.name === "extra-skill")).toBeDefined(); }); + test("marketplace-root skills manifest field replaces default skills directory", async () => { + // Claude path-behavior rules carve out marketplace entries whose source is the + // marketplace root: their manifest `skills` field selects the published + // subdirectories instead of also loading the root `skills/` directory. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-marketplace-root"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "skills", "unpublished-root-skill"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "plugins", "published", "skills", "published-skill"), { + recursive: true, + }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-marketplace-root@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, "marketplace.json"), + JSON.stringify({ + name: "market", + owner: { name: "Market" }, + plugins: [{ name: "manifest-skills-marketplace-root", source: "./" }], + }), + ); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./plugins/published/skills"] }), + ); + await fs.writeFile( + path.join(pluginPath, "skills", "unpublished-root-skill", "SKILL.md"), + "---\nname: unpublished-root-skill\ndescription: Unpublished root skill\n---\nBody\n", + ); + await fs.writeFile( + path.join(pluginPath, "plugins", "published", "skills", "published-skill", "SKILL.md"), + "---\nname: published-skill\ndescription: Published skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "published-skill")).toBeDefined(); + expect(result.all.find(s => s.name === "unpublished-root-skill")).toBeUndefined(); + }); + test("array-form skills entry pointing at a directory containing SKILL.md loads the single skill", async () => { // Per Claude plugins reference: a skills path may point directly at a directory whose // SKILL.md is the skill (frontmatter name → invocation, directory basename → fallback). From 92d3286a41419d34d4c25b5737c58f7ff8653739 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:06:48 +0200 Subject: [PATCH 81/91] fix(coding-agent): refresh advisor on routed role reloads --- packages/coding-agent/src/config/settings.ts | 2 + .../coding-agent/src/session/agent-session.ts | 2 +- .../coding-agent/test/advisor-toggle.test.ts | 76 ++++++++++++++++++- 3 files changed, 78 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index 268b147f7..c8aca9a30 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -483,11 +483,13 @@ export class Settings { async reloadForCwd(cwd: string): Promise { const normalized = path.normalize(cwd); if (normalized === this.#cwd) return; + const prevModelRoles = this.get("modelRoles"); this.#cwd = normalized; if (this.#persist) { this.#project = await this.#loadProjectSettings(); } this.#rebuildMerged(); + this.#fireEffectiveSettingChanged("modelRoles", this.get("modelRoles"), prevModelRoles); this.#fireAllHooks(); } diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 687874031..4097e5ccf 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2413,7 +2413,7 @@ export class AgentSession { #advisorRuntimeSignature(config: AdvisorConfig, slug: string, model: Model, thinkingLevel: ThinkingLevel): string { const tools = config.tools?.length ? config.tools.join("\u001e") : ""; const instructions = config.instructions?.trim() ?? ""; - return [config.name, slug, model.provider, model.id, thinkingLevel, tools, instructions].join("\u001f"); + return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join("\u001f"); } #advisorRuntimeMatchesCurrentConfig(): boolean { diff --git a/packages/coding-agent/test/advisor-toggle.test.ts b/packages/coding-agent/test/advisor-toggle.test.ts index 0d911aad0..b22067375 100644 --- a/packages/coding-agent/test/advisor-toggle.test.ts +++ b/packages/coding-agent/test/advisor-toggle.test.ts @@ -1,4 +1,5 @@ import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; import * as path from "node:path"; import { Agent, type AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { Model } from "@oh-my-pi/pi-ai"; @@ -6,9 +7,10 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; -import { TempDir } from "@oh-my-pi/pi-utils"; +import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils"; describe("AgentSession advisor toggle", () => { let sharedDir: TempDir; @@ -22,6 +24,7 @@ describe("AgentSession advisor toggle", () => { authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db")); authStorage.setRuntimeApiKey("anthropic", "test-key"); authStorage.setRuntimeApiKey("openai", "test-key"); + authStorage.setRuntimeApiKey("openrouter", "test-key"); modelRegistry = new ModelRegistry(authStorage); const bundled = getBundledModel("anthropic", "claude-sonnet-4-5"); const replacement = getBundledModel("openai", "gpt-4o-mini"); @@ -110,6 +113,77 @@ describe("AgentSession advisor toggle", () => { expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); }); + it("refreshes the live advisor when only the advisor route changes", () => { + session.settings.setModelRole("advisor", "openrouter/z-ai/glm-4.7@cerebras"); + expect(session.setAdvisorEnabled(true)).toBe(true); + expect(session.getAdvisorAgent()?.state.model.provider).toBe("openrouter"); + expect(session.getAdvisorAgent()?.state.model.id).toBe("z-ai/glm-4.7"); + expect( + (session.getAdvisorAgent()?.state.model.compat as { openRouterRouting?: { only?: string[] } } | undefined) + ?.openRouterRouting?.only, + ).toEqual(["cerebras"]); + + session.settings.setModelRole("advisor", "openrouter/z-ai/glm-4.7@fireworks"); + + expect(session.getAdvisorAgent()?.state.model.provider).toBe("openrouter"); + expect(session.getAdvisorAgent()?.state.model.id).toBe("z-ai/glm-4.7"); + expect( + (session.getAdvisorAgent()?.state.model.compat as { openRouterRouting?: { only?: string[] } } | undefined) + ?.openRouterRouting?.only, + ).toEqual(["fireworks"]); + }); + + it("refreshes the live advisor after project model-role reloads", async () => { + const projectA = path.join(tempDir.path(), "project-a"); + const projectB = path.join(tempDir.path(), "project-b"); + const agentDir = path.join(tempDir.path(), "agent"); + fs.mkdirSync(getProjectAgentDir(projectA), { recursive: true }); + fs.mkdirSync(getProjectAgentDir(projectB), { recursive: true }); + fs.mkdirSync(agentDir, { recursive: true }); + await Bun.write( + path.join(getProjectAgentDir(projectA), "settings.json"), + JSON.stringify({ modelRoles: { advisor: `${model.provider}/${model.id}` } }), + ); + await Bun.write( + path.join(getProjectAgentDir(projectB), "settings.json"), + JSON.stringify({ modelRoles: { advisor: `${replacementModel.provider}/${replacementModel.id}` } }), + ); + + const settings = await Settings.loadIsolated({ + cwd: projectA, + agentDir, + overrides: { "compaction.enabled": false }, + }); + const customSession = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }), + sessionManager: SessionManager.create(tempDir.path(), tempDir.path()), + settings, + modelRegistry, + advisorTools: [], + }); + + try { + expect(customSession.setAdvisorEnabled(true)).toBe(true); + expect(customSession.getAdvisorAgent()?.state.model.provider).toBe(model.provider); + expect(customSession.getAdvisorAgent()?.state.model.id).toBe(model.id); + + await settings.reloadForCwd(projectB); + + expect(customSession.getAdvisorAgent()?.state.model.provider).toBe(replacementModel.provider); + expect(customSession.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); + } finally { + await customSession.dispose(); + AgentStorage.resetInstance(); + } + }); + it("keeps explicit enable idempotent when the advisor config is unchanged", () => { session.settings.setModelRole("advisor", `${model.provider}/${model.id}`); expect(session.setAdvisorEnabled(true)).toBe(true); From f313669705bce716b2791770d8af2828f99be476 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:05:53 +0200 Subject: [PATCH 82/91] fix(agent): respect vetoed auto-handoff hooks --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/session/agent-session.ts | 8 +- .../test/agent-session-handoff.test.ts | 87 +++++++++++++++++++ 3 files changed, 95 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 39013c137..42ef42052 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -207,7 +207,7 @@ - Fixed Windows session tail loss after atomic compaction rewrites by fencing append writers during full-file replacement and gating the atomic publish on a `commitGuard` that the storage backend checks synchronously before rename, so a concurrent `flushSync` (Ctrl+C / session-exit) is not overwritten by the stale body serialized before it ran. Covers post-compaction prompts, tool results, title changes, and exit diagnostics on the current JSONL path ([#4338](https://github.com/can1357/oh-my-pi/issues/4338)). ### Fixed -- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting `session_switch` with `reason: "handoff"` before replacing the outgoing session ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). +- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). ## [16.3.4] - 2026-07-03 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index be6d8cad6..327eac4b9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -882,6 +882,7 @@ export interface HandoffResult { export interface SessionHandoffOptions { autoTriggered?: boolean; signal?: AbortSignal; + onSwitchCancelled?: () => void; } /** Result from cycleModel() */ @@ -10091,6 +10092,7 @@ export class AgentSession { })) as SessionBeforeSwitchResult | undefined; if (result?.cancel) { + options?.onSwitchCancelled?.(); return undefined; } } @@ -12373,13 +12375,17 @@ export class AgentSession { // queue, not the core steering queue (which handoff's agent.reset() would wipe). await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action }); if (action === "handoff") { + let handoffSwitchCancelled = false; const handoffFocus = AUTO_HANDOFF_THRESHOLD_FOCUS; const handoffResult = await this.handoff(handoffFocus, { autoTriggered: true, signal: autoCompactionSignal, + onSwitchCancelled: () => { + handoffSwitchCancelled = true; + }, }); if (!handoffResult) { - const aborted = autoCompactionSignal.aborted; + const aborted = autoCompactionSignal.aborted || handoffSwitchCancelled; if (aborted) { await this.#emitSessionEvent({ type: "auto_compaction_end", diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index 5e1229aba..8b00a4e1c 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -1682,6 +1682,93 @@ describe("AgentSession handoff", () => { }); }); + it("treats a vetoed auto-handoff switch as cancelled instead of falling back", async () => { + session.settings.set("compaction.strategy", "handoff"); + session.settings.set("compaction.thresholdPercent", 1); + session.settings.set("contextPromotion.enabled", false); + + const model = session.model; + if (!model) { + throw new Error("Expected model to be set"); + } + + const extensionsResult = await loadExtensions([], tempDir.path()); + const extensionRunner = new ExtensionRunner( + extensionsResult.extensions, + extensionsResult.runtime, + tempDir.path(), + sessionManager, + modelRegistry, + ); + vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch"); + const emit = extensionRunner.emit.bind(extensionRunner); + const emitSpy = vi.spyOn(extensionRunner, "emit").mockImplementation(event => { + if (event.type === "session_before_switch") { + return { cancel: true }; + } + return emit(event); + }); + + await session.dispose(); + session = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }), + sessionManager, + settings: session.settings, + modelRegistry, + extensionRunner, + obfuscator, + }); + session.subscribe(event => { + events.push(event); + }); + const previousSessionFile = session.sessionFile; + const generateHandoffSpy = vi + .spyOn(compactionModule, "generateHandoffFromContext") + .mockResolvedValue("## Goal\nContinue from here"); + const assistantMessage: AssistantMessage = { + role: "assistant", + content: [{ type: "text", text: "maintenance trigger" }], + api: model.api, + provider: model.provider, + model: model.id, + stopReason: "stop", + usage: { + input: 10_000, + output: 1_000, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 11_000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: Date.now(), + }; + + session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); + await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1); + + expect(generateHandoffSpy).toHaveBeenCalledTimes(1); + expect(emitSpy).toHaveBeenCalledWith({ type: "session_before_switch", reason: "handoff" }); + expect(emitSpy).not.toHaveBeenCalledWith(expect.objectContaining({ type: "session_switch" })); + expect(session.sessionFile).toBe(previousSessionFile); + expect(sessionManager.getEntries().filter(entry => entry.type === "compaction")).toHaveLength(0); + const endEvents = events.filter(event => event.type === "auto_compaction_end"); + expect(endEvents).toHaveLength(1); + expect(endEvents[0]).toMatchObject({ + type: "auto_compaction_end", + action: "handoff", + aborted: true, + willRetry: false, + }); + }); + it("resets to the base system prompt before generating a handoff", async () => { const model = session.model; if (!model) { From 92b2923f7bf81a34bb4e91448f31ce7ad27c2c13 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:09:19 +0200 Subject: [PATCH 83/91] fix(coding-agent): normalize fallback chain picker writes --- .../modes/controllers/selector-controller.ts | 2 +- .../selector-settings-side-effects.test.ts | 73 +++++++++++++++++++ 2 files changed, 74 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index 0333c3d15..0c549ca44 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -604,7 +604,7 @@ export class SelectorController { if (action === "retryFallback" && role !== null) { const fallbackSelector = formatModelSelectorValue(selectorValue, concreteThinking); const fallbackChains = this.ctx.settings.get("retry.fallbackChains"); - const chain = fallbackChains[role] ?? []; + const chain = Array.isArray(fallbackChains[role]) ? fallbackChains[role] : []; this.ctx.settings.set("retry.fallbackChains", { ...fallbackChains, [role]: [fallbackSelector, ...chain.filter(existing => existing !== fallbackSelector)], diff --git a/packages/coding-agent/test/selector-settings-side-effects.test.ts b/packages/coding-agent/test/selector-settings-side-effects.test.ts index 3e54ea139..8a16e03d8 100644 --- a/packages/coding-agent/test/selector-settings-side-effects.test.ts +++ b/packages/coding-agent/test/selector-settings-side-effects.test.ts @@ -1,6 +1,9 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { stripVTControlCharacters } from "node:util"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller"; +import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { beginSettingsTest, restoreSettingsTestState, type SettingsTestState } from "./helpers/settings-test-state"; let settingsState: SettingsTestState | undefined; @@ -51,4 +54,74 @@ describe("selector setting side effects", () => { expect(invalidate).toHaveBeenCalledTimes(1); expect(requestRender).toHaveBeenCalledTimes(1); }); + + it("replaces malformed default retry fallback chains from the model selector action", async () => { + const testTheme = await getThemeByName("dark"); + if (!testTheme) throw new Error("Failed to load dark theme for model selector test"); + setThemeInstance(testTheme); + + const settings = Settings.isolated({}); + settings.set("retry.fallbackChains", { default: "not-an-array" } as unknown as Record); + const fallback = buildModel({ + id: "retry-fallback-model", + name: "retry-fallback-model", + api: "ollama-chat", + baseUrl: "https://example.com", + reasoning: false, + provider: "test", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 1024, + }); + const showStatus = vi.fn(); + const showError = vi.fn(); + const controller = new SelectorController({ + ui: { requestRender: vi.fn(), setFocus: vi.fn() }, + editorContainer: { clear: vi.fn(), addChild: vi.fn() }, + editor: {}, + settings, + session: { + model: undefined, + modelRegistry: { + getAll: () => [fallback], + getDiscoverableProviders: () => [], + }, + scopedModels: [{ model: fallback }], + getContextUsage: () => undefined, + }, + statusLine: { invalidate: vi.fn() }, + updateEditorBorderColor: vi.fn(), + keybindings: { getKeys: () => [] }, + showStatus, + showError, + } as unknown as ConstructorParameters[0]); + let selector: { handleInput(input: string): void; render(width: number): string[] } | undefined; + controller.showSelector = create => { + const result = create(() => {}); + selector = result.component as typeof selector; + }; + + controller.showModelSelector(); + if (!selector) throw new Error("Expected model selector to be shown"); + selector.handleInput("\n"); + for (let attempt = 0; attempt < 20; attempt++) { + const selectedLine = stripVTControlCharacters(selector.render(220).join("\n")) + .split("\n") + .find(line => { + if (!line.includes("Set as DEFAULT retry fallback")) return false; + const trimmed = line.trimStart(); + return trimmed.startsWith("❯") || trimmed.startsWith("▸") || trimmed.startsWith(">"); + }); + if (selectedLine) break; + selector.handleInput("\x1b[B"); + if (attempt === 19) throw new Error("Default retry fallback action was not selectable"); + } + selector.handleInput("\n"); + await Promise.resolve(); + + expect(showError).not.toHaveBeenCalled(); + expect(settings.get("retry.fallbackChains")).toEqual({ default: ["test/retry-fallback-model"] }); + expect(showStatus).toHaveBeenCalledWith("Default fallback model: test/retry-fallback-model"); + }); }); From 10745dfc94b61fbdbbee28f1197bf9d7e08ff4a3 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:04:14 +0200 Subject: [PATCH 84/91] fix(tui): tighten slash autocomplete test --- packages/tui/test/editor.test.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index da3f0f7d3..1e0b27dfb 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -324,8 +324,11 @@ describe("Editor component", () => { ), ); + const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers(); + editor.onAutocompleteUpdate = resolveAutocompleteUpdated; + editor.handleInput("/"); - await Bun.sleep(0); + await autocompleteUpdated; const rendered = editor.render(80).map(line => stripVTControlCharacters(line)); for (let i = 0; i < 10; i += 1) { From e2a01aa4f47aeeaf1bd8dc5a4688ab383bd3def7 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 14:56:14 +0200 Subject: [PATCH 85/91] fix(catalog): preserve LiteLLM rich metadata --- .../src/provider-models/openai-compat.ts | 39 ++++++++++++++++--- .../catalog/test/litellm-provider.test.ts | 15 +++---- 2 files changed, 42 insertions(+), 12 deletions(-) diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index c1e2a046b..83cd58136 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3036,6 +3036,10 @@ type LiteLLMRichEndpointModel = { model: ModelSpec; supportsVision: unknown; supportsReasoning: unknown; + hasContextWindow: boolean; + hasMaxTokens: boolean; + hasToolMetadata: boolean; + hasSupportedOpenAIParams: boolean; }; const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const; @@ -3307,13 +3311,23 @@ async function fetchLiteLLMRichEndpoint( const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl); if (model) { const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision"); + const supportsReasoning = getLiteLLMMetadataValue(entry, "supports_reasoning"); + const supportsFunctionCalling = getLiteLLMMetadataValue(entry, "supports_function_calling"); + const supportedOpenAIParams = getSupportedOpenAIParams(entry); if (supportsVision !== true && supportsVision !== false) { incompleteVisionMetadata = true; } deduped.set(model.id, { model, supportsVision, - supportsReasoning: getLiteLLMMetadataValue(entry, "supports_reasoning"), + supportsReasoning, + hasContextWindow: toPositiveNumber(getLiteLLMMetadataValue(entry, "max_input_tokens"), null) !== null, + hasMaxTokens: toPositiveNumber(getLiteLLMMetadataValue(entry, "max_output_tokens"), null) !== null, + hasToolMetadata: + supportsFunctionCalling === true || + supportsFunctionCalling === false || + supportedOpenAIParams !== undefined, + hasSupportedOpenAIParams: supportedOpenAIParams !== undefined, }); } } @@ -3341,25 +3355,40 @@ export async function fetchLiteLLMRichModels( if (!result) { continue; } + const hadPriorModels = deduped.size > 0; for (const next of result.models) { const existing = deduped.get(next.model.id); if (!existing) { - deduped.set(next.model.id, next); + if (!hadPriorModels) { + deduped.set(next.model.id, next); + } continue; } - const model = { + const model: ModelSpec = { ...existing.model, - ...next.model, name: next.model.name === next.model.id ? existing.model.name : next.model.name, + contextWindow: next.hasContextWindow ? next.model.contextWindow : existing.model.contextWindow, + maxTokens: next.hasMaxTokens ? next.model.maxTokens : existing.model.maxTokens, input: next.supportsVision === true || next.supportsVision === false ? next.model.input : existing.model.input, reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning, + compat: next.hasSupportedOpenAIParams ? next.model.compat : existing.model.compat, }; + if (next.hasToolMetadata) { + model.supportsTools = next.model.supportsTools; + } deduped.set(next.model.id, { ...next, model }); } - if (!result.incompleteVisionMetadata) { + let hasIncompleteVisionMetadata = false; + for (const entry of deduped.values()) { + if (entry.supportsVision !== true && entry.supportsVision !== false) { + hasIncompleteVisionMetadata = true; + break; + } + } + if (!hasIncompleteVisionMetadata) { break; } } diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index 2524d014d..c353ac16f 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -490,14 +490,16 @@ describe("LiteLLM provider discovery", () => { { model_group: "vision-proxy-model", model_name: "Vision Proxy Model", - max_input_tokens: 128_000, - max_output_tokens: 16_000, + max_input_tokens: 64_000, + max_output_tokens: 8_000, }, ], }); } if (url === "http://primary:4000/v2/model/info") { - return new Response("{}", { status: 404 }); + return Response.json({ + data: [{ model_name: "unrelated-v2-model", model_info: { supports_vision: false } }], + }); } if (url === "http://primary:4000/model/info") { return Response.json({ @@ -506,8 +508,6 @@ describe("LiteLLM provider discovery", () => { { model_name: "vision-proxy-model", model_info: { - max_input_tokens: 128_000, - max_output_tokens: 16_000, supports_vision: true, }, }, @@ -531,12 +531,13 @@ describe("LiteLLM provider discovery", () => { expect(calls).toContain("http://primary:4000/v2/model/info"); expect(calls).toContain("http://primary:4000/model/info"); expect(calls).not.toContain("http://primary:4000/v1/models"); + expect(models).toHaveLength(1); expect(models?.find(model => model.id === "vision-proxy-model")).toMatchObject({ id: "vision-proxy-model", name: "Vision Proxy Model", input: ["text", "image"], - contextWindow: 128_000, - maxTokens: 16_000, + contextWindow: 64_000, + maxTokens: 8_000, }); }); From 6af3c8080e4095825bbaeb60cb1a796fb2eea51c Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:04:39 +0200 Subject: [PATCH 86/91] fix(tui): preserve description-matched skill completion --- packages/tui/src/components/editor.ts | 4 +++- .../test/editor-autocomplete-actions.test.ts | 21 +++++++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 492aafd8d..bcaba89a8 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -2914,7 +2914,9 @@ export class Editor implements Component, Focusable { // when the current query would still surface it. `tmp` after a bare // slash therefore falls through to file completion instead of // rewriting the user's `/tmp` to `/skill:…`. - if (scoreCommandTextMatch(token.slice(1).toLowerCase(), item.value.toLowerCase()) > 0) return true; + const lowerToken = token.slice(1).toLowerCase(); + if (scoreCommandTextMatch(lowerToken, item.value.toLowerCase()) > 0) return true; + if (item.description && scoreCommandTextMatch(lowerToken, item.description.toLowerCase()) > 0) return true; } } return false; diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index 119fbf807..35b5becfa 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -272,6 +272,27 @@ describe("Editor Enter handler sync slash completion", () => { expect(editor.isShowingAutocomplete()).toBe(false); }); + it("accepts a stale mid-prompt skill suggestion when the live token still matches its description", async () => { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider( + [ + { name: "skill:hardening", description: "Security scan" }, + { name: "model", description: "Switch model" }, + ], + "/tmp", + ), + ); + + await openMidPromptSkillAutocomplete(editor, "run a "); + // Race the 100 ms debounce: type a query that matches only the skill description. + editor.handleInput("scan"); + editor.handleInput("\t"); + + expect(editor.getText()).toBe("run a /skill:hardening "); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + it("opens mid-prompt skill autocomplete and inserts the skill token without wiping the draft on Tab", async () => { const editor = new Editor(defaultEditorTheme); editor.setAutocompleteProvider( From 8b6e4cb03a184ff461fba53b9949a88281c5674f Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 14:44:52 +0200 Subject: [PATCH 87/91] fix(coding-agent): validate live github ref completions --- .../test/modes/github-ref-autocomplete.test.ts | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts index ba77380fa..0cd330b42 100644 --- a/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts +++ b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts @@ -132,6 +132,21 @@ describe("github-ref autocomplete — provider integration", () => { expect(result.lines).toEqual(["review pr://3164 "]); }); + it("revalidates stale prefixes against the live cursor token before applying", async () => { + const provider = makeProvider(); + const staleSuggestions = await provider.getSuggestions(["review #316"], 0, 11); + expect(staleSuggestions).not.toBeNull(); + const stalePr = staleSuggestions!.items[0]!; + + const updatedNumber = provider.applyCompletion(["review #3164"], 0, 12, stalePr, staleSuggestions!.prefix); + expect(updatedNumber.lines).toEqual(["review pr://3164 "]); + expect(updatedNumber.cursorCol).toBe("review pr://3164 ".length); + + const embeddedHash = provider.applyCompletion(["owner/repo#3164"], 0, 15, stalePr, staleSuggestions!.prefix); + expect(embeddedHash.lines).toEqual(["owner/repo#3164"]); + expect(embeddedHash.cursorCol).toBe(15); + }); + it("does not offer candidates for embedded hashes (falls through to other providers)", async () => { const provider = makeProvider(); const isRef = (value: string) => value.startsWith("pr://") || value.startsWith("issue://"); From 1bb29873ea1537f0cd99629de15b936992941409 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:34:37 +0200 Subject: [PATCH 88/91] fix(agent): adapted scoped TTSR abort labels to completed-call retention - derive per-tool abort labels from a tool-scoped abort signal for provider-built aborted messages - restore main's single-call TTSR label test dropped by the merge - complete the innocent read in the sibling-label test; incomplete matched calls mint no placeholder under the retention policy --- packages/agent/src/agent-loop.ts | 8 +- .../test/agent-session-concurrent.test.ts | 129 +++++++++++++++++- 2 files changed, 129 insertions(+), 8 deletions(-) diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index cdfa20a73..a69fdf2e3 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -939,9 +939,15 @@ async function runLoopBody( (c): c is ToolCallContent => c.type === "toolCall" && (c as CursorExecResolvedCarrier)[kCursorExecResolved] !== true, ); + // Provider-built aborted messages (stream error events) carry no + // per-tool labels; derive them from a tool-scoped abort signal so + // only the matching call is blamed and siblings stay neutral. + const scopedAbort = toolScopedAbortReason(signal); + const toolCallAbortMessages = + message.toolCallAbortMessages ?? (scopedAbort ? buildToolCallAbortMessages(message, scopedAbort) : undefined); const toolResults: ToolResultMessage[] = []; for (const toolCall of toolCalls) { - const errorMessage = message.toolCallAbortMessages?.[toolCall.id] ?? message.errorMessage; + const errorMessage = toolCallAbortMessages?.[toolCall.id] ?? message.errorMessage; const result = createAbortedToolResult(toolCall, stream, message.stopReason, errorMessage); currentContext.messages.push(result); newMessages.push(result); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 889fe8880..696732316 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -1124,6 +1124,117 @@ describe("AgentSession TTSR resume gate", () => { expect(session.isStreaming).toBe(false); }); + it("labels aborted tool placeholders with the TTSR rule reason", async () => { + collapseSchedulerSettleDelays(); + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + let streamCallCount = 0; + + const ttsrManager = new TtsrManager({ + enabled: true, + contextMode: "discard", + interruptMode: "always", + repeatMode: "once", + repeatGap: 10, + }); + ttsrManager.addRule(testRule); + + const toolCallContent: ToolCall = { + type: "toolCall", + id: "call_ttsr_abort_reason", + name: "mock_edit", + arguments: { snippet: "let val = result.unwrap(" }, + }; + + const makeToolCallMsg = (stopReason: "toolUse" | "aborted" = "toolUse"): AssistantMessage => ({ + role: "assistant", + content: [toolCallContent], + api: "anthropic-messages", + provider: "anthropic", + model: "mock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason, + timestamp: Date.now(), + }); + + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [] }, + streamFn: (_model, _context, options) => { + streamCallCount++; + const stream = new AssistantMessageEventStream(); + const signal = options?.signal; + if (streamCallCount === 1) { + queueMicrotask(() => { + const partial = makeToolCallMsg(); + if (signal) { + signal.addEventListener( + "abort", + () => { + stream.push({ + type: "error", + reason: "aborted", + error: makeToolCallMsg("aborted"), + }); + }, + { once: true }, + ); + } + stream.push({ type: "start", partial }); + stream.push({ type: "toolcall_start", contentIndex: 0, partial }); + stream.push({ + type: "toolcall_delta", + contentIndex: 0, + delta: 'let val = result.unwrap("oops")', + partial, + }); + // The TTSR abort placeholder is only minted for tool calls that reached + // `toolcall_end`: the agent loop drops incomplete tool calls from an + // aborted turn (partial args are unsafe to replay). Complete the call + // before the rule-driven abort fires so the labeled placeholder survives. + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: toolCallContent, partial }); + }); + } else { + pushContinuationStream(stream, () => {}); + } + return stream; + }, + }); + + const sessionManager = SessionManager.inMemory(); + const settings = Settings.isolated(); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-abort-reason.db")); + authStorages.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + session = new AgentSession({ agent, sessionManager, settings, modelRegistry, ttsrManager }); + + await session.prompt("Write some Rust code"); + + const toolResult = sessionManager + .getEntries() + .find( + entry => + entry.type === "message" && + entry.message.role === "toolResult" && + entry.message.toolCallId === toolCallContent.id, + ); + expect(toolResult?.type).toBe("message"); + const text = + toolResult?.type === "message" && toolResult.message.role === "toolResult" + ? (toolResult.message.content.find((part): part is { type: "text"; text: string } => part.type === "text") + ?.text ?? "") + : ""; + expect(text).toContain("Tool execution was aborted: TTSR matched rule: no-unwrap"); + expect(text).not.toContain("Request was aborted"); + }); + it("labels only the matching aborted tool placeholder with the TTSR rule reason", async () => { collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; @@ -1200,11 +1311,12 @@ describe("AgentSession TTSR resume gate", () => { delta: 'let val = result.unwrap("oops")', partial, }); - // The TTSR abort placeholder is only minted for tool calls that reached + // The abort placeholder is only minted for tool calls that reached // `toolcall_end`: the agent loop drops incomplete tool calls from an - // aborted turn (partial args are unsafe to replay). Complete the call - // before the rule-driven abort fires so the labeled placeholder survives. - stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: toolCallContent, partial }); + // aborted turn (partial args are unsafe to replay). Complete the + // innocent read before the rule-driven abort fires so its placeholder + // survives and can carry the neutral sibling label. + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: readToolCallContent, partial }); }); } else { pushContinuationStream(stream, () => {}); @@ -1234,13 +1346,16 @@ describe("AgentSession TTSR resume gate", () => { ?.content.find((part): part is { type: "text"; text: string } => part.type === "text")?.text ?? ""; const readText = toolResultText(readToolCallContent.id); - const matchedText = toolResultText(matchedToolCallContent.id); expect(readText).toContain("Tool execution was aborted: TTSR interrupt on another tool call"); expect(readText).not.toContain("TTSR matched rule: no-unwrap"); - expect(matchedText).toContain("Tool execution was aborted: TTSR matched rule: no-unwrap"); - expect(matchedText).not.toContain("Request was aborted"); + // The matching call never reached `toolcall_end`, so the loop drops it from + // the aborted turn (partial args are unsafe to replay) and no placeholder is + // minted. The rule label for a completed matching call is covered by the + // single-call test above. + expect(toolResultText(matchedToolCallContent.id)).toBe(""); }); + it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; From 3ee194dcc7db43ac3eb6508e7bdcbc3aca99a6fd Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:35:00 +0200 Subject: [PATCH 89/91] chore: normalized changelog entries for merged pull requests --- packages/agent/CHANGELOG.md | 8 +-- packages/ai/CHANGELOG.md | 21 ++++---- packages/catalog/CHANGELOG.md | 8 +-- packages/coding-agent/CHANGELOG.md | 87 ++++++++++-------------------- packages/tui/CHANGELOG.md | 8 +-- packages/utils/CHANGELOG.md | 1 + 6 files changed, 49 insertions(+), 84 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 16dec65eb..0d1b46a12 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added per-tool abort metadata so stream-wide aborts can label matching tool-call placeholders separately from unaffected sibling calls ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). + ### Fixed - Fixed handoff generation retrying with `toolChoice: "auto"` when custom OpenAI-compatible providers reject `toolChoice: "none"` with an auto-only 400. ([#4715](https://github.com/can1357/oh-my-pi/issues/4715)) @@ -202,10 +206,6 @@ - Fixed `PI_DIALECT=minimax` being ignored by the owned tool-calling env selector. ([#2759](https://github.com/can1357/oh-my-pi/issues/2759)) - -### Added - -- Added per-tool abort metadata so stream-wide aborts can label matching tool-call placeholders separately from unaffected sibling calls ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.0.1] - 2026-06-15 ### Fixed diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 757488a84..a72f0fc1e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,13 +2,21 @@ ## [Unreleased] +### Added + +- Added `AssistantMessage.toolCallAbortMessages` for per-tool placeholder labels on aborted assistant turns ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). + ### Fixed - Fixed Anthropic replay 400s (`tool_use ids were found without tool_result blocks immediately after`) when a persisted assistant turn carries content after a completed tool call — such as a mid-turn `server-side-fallback` handoff (fallback block plus continued text/tool calls after the primary model's `tool_use`) or trailing text from cross-provider replays — by stable-partitioning assistant content so all `tool_use` blocks trail the non-`tool_use` chain. ([#4781](https://github.com/can1357/oh-my-pi/issues/4781), [#544](https://github.com/can1357/oh-my-pi/issues/544)) - Fixed access-token-only OAuth credentials attempting token refresh with an empty refresh token after expiry. - Fixed gateway usage-limit retries falling through to cross-provider model fallback before trying a sibling credential from the same provider. - Fixed Codex usage-limit rotation treating Plus and K-12 accounts as separate quota groups for shared 5-hour/7-day windows. - +- Fixed OpenAI Responses streams that end with `response.done` being misclassified as premature stream closures. +- Fixed OpenCode Go `/login` credentials being shadowed by an existing `OPENCODE_API_KEY` env fallback after switching accounts. ([#4688](https://github.com/can1357/oh-my-pi/issues/4688)) +- Fixed OpenAI Codex WebSocket continuations to treat proxy stale-anchor codes such as `codex_previous_response_stale` as an expired `previous_response_id` chain — same recovery class as the OpenAI-standard `previous_response_not_found` — so the turn is retried with full context instead of surfacing the error to the user ([#4624](https://github.com/can1357/oh-my-pi/issues/4624)). +- Fixed Azure Foundry Anthropic utility requests to omit the structured-output beta whenever strict tools are disabled, preventing `structured_outputs not supported in your workspace` failures for Sonnet 5 compaction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). +- Fixed OAuth `launchUrl` advertisement for flows whose redirect never returns to the local callback server: custom-scheme redirects (e.g. GitLab Duo's `vscode://` URI, which `new URL` parses without complaint) and fixed non-loopback hosts no longer receive a `http://localhost:/launch` copy target that misrepresents the callback endpoint and resolves nowhere for remote users. ## [16.3.11] - 2026-07-06 @@ -16,10 +24,6 @@ - Fixed `openai-codex-responses` fresh plan execution requests that contained only system/developer guidance by mirroring the final instruction as user input so Codex accepts the first turn. ([#4714](https://github.com/can1357/oh-my-pi/issues/4714)) - Fixed Codex WebSocket compact/resume delta diagnostics to record request shape and raw-vs-displayed usage buckets, so persistent server-reported uncached suffixes without `orchestration_*` fields are visible in debug stats. ([#4707](https://github.com/can1357/oh-my-pi/issues/4707)) -### Fixed - -- Fixed OpenAI Responses streams that end with `response.done` being misclassified as premature stream closures. -- Fixed OpenCode Go `/login` credentials being shadowed by an existing `OPENCODE_API_KEY` env fallback after switching accounts. ([#4688](https://github.com/can1357/oh-my-pi/issues/4688)) ## [16.3.10] - 2026-07-06 @@ -27,9 +31,6 @@ - Fixed Ollama/Ollama Cloud EOS-only completions to retry empty stops with a single output token before the agent loop can halt silently. ([#4659](https://github.com/can1357/oh-my-pi/issues/4659)) - Fixed Claude Sonnet 5 failing every request on feature-gated gateways (Azure Foundry, OpenAI-compatible relays) that reject strict tools with "structured_outputs not supported" — the rejection is now classified as a strict-tool rejection, so the request retries without strict tools and the session remembers the downgrade. -- Fixed OpenAI Codex WebSocket continuations to treat proxy stale-anchor codes such as `codex_previous_response_stale` as an expired `previous_response_id` chain — same recovery class as the OpenAI-standard `previous_response_not_found` — so the turn is retried with full context instead of surfacing the error to the user ([#4624](https://github.com/can1357/oh-my-pi/issues/4624)). -- Fixed Azure Foundry Anthropic utility requests to omit the structured-output beta whenever strict tools are disabled, preventing `structured_outputs not supported in your workspace` failures for Sonnet 5 compaction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). -- Fixed OAuth `launchUrl` advertisement for flows whose redirect never returns to the local callback server: custom-scheme redirects (e.g. GitLab Duo's `vscode://` URI, which `new URL` parses without complaint) and fixed non-loopback hosts no longer receive a `http://localhost:/launch` copy target that misrepresents the callback endpoint and resolves nowhere for remote users. ## [16.3.7] - 2026-07-05 @@ -695,10 +696,6 @@ - Fixed OpenAI-compatible Ollama completions that return empty `finish_reason:length` after filling `num_ctx` so they surface an actionable context-window error instead of an empty length stop. ([#2774](https://github.com/can1357/oh-my-pi/issues/2774)) - Fixed Codex browser login issuing credentials for the `opencode` OAuth originator while OMP requests identify as `pi`, which could make the first authenticated Codex request return 401 ([#2696](https://github.com/can1357/oh-my-pi/issues/2696)). - -### Added - -- Added `AssistantMessage.toolCallAbortMessages` for per-tool placeholder labels on aborted assistant turns ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.0.1] - 2026-06-15 ### Added diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 72dabb0b5..1a020e17d 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -5,6 +5,8 @@ ### Fixed - Fixed LiteLLM discovery stopping at `/model_group/info` when that endpoint omitted `supports_vision`; it now continues to `/model/info` and preserves `model_info.supports_vision=true` for vision-capable proxy models. ([#4747](https://github.com/can1357/oh-my-pi/issues/4747)) +- Fixed LiteLLM discovery to fall back to bundled catalog metadata when `models.dev` lacks a model reference, preserving reasoning and thinking support for models such as `glm-5.2`. ([#4695](https://github.com/can1357/oh-my-pi/issues/4695)) +- Detected Azure AI Inference / Foundry Anthropic routes as strict-tool-incompatible so resolved Anthropic compat disables strict tools before request construction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). ## [16.3.11] - 2026-07-06 @@ -18,18 +20,12 @@ - Updated naming format for various synthetic models to include provider prefix - Adjusted context window limit for MiniMax-M3 model - Updated pricing for select models -### Fixed - -- Fixed LiteLLM discovery to fall back to bundled catalog metadata when `models.dev` lacks a model reference, preserving reasoning and thinking support for models such as `glm-5.2`. ([#4695](https://github.com/can1357/oh-my-pi/issues/4695)) ## [16.3.10] - 2026-07-06 ### Fixed - Fixed LiteLLM rich discovery to ignore unusable sentinel placeholders and continue to `/v2/model/info` for real models. ([#4655](https://github.com/can1357/oh-my-pi/issues/4655)) -### Fixed - -- Detected Azure AI Inference / Foundry Anthropic routes as strict-tool-incompatible so resolved Anthropic compat disables strict tools before request construction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). ## [16.3.9] - 2026-07-06 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7591620f8..d10d49884 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Typing `#` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline. Naming the type (`pr #3164` / `issue #3164`) constrains the candidates to that kind, and embedded hashes like `owner/repo#N`, `foo#N`, or URL fragments are left untouched ([#3218](https://github.com/can1357/oh-my-pi/issues/3218)) + ### Changed - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. @@ -10,20 +14,40 @@ - Improved handling of unawaited promises in JS eval cells to prevent process crashes - Added warning logs for unhandled rejections originating from finished eval cells - - Improved advisor robustness by blocking exhausted accounts during consecutive turn failures - Fixed advisor turns hammering the same usage-limited account: a failed advisor turn now marks the exhausted credential blocked (with the provider's retry hint and usage-report reset time), so the next retry rotates to a sibling instead of re-picking the blocked account every few seconds. Previously the in-stream auth retry rotated within a request but never blocked the last failing credential, and the advisor loop — unlike the primary retry pipeline — never called `markUsageLimitReached`. - Added the account key to the `codex-auto-reset: skipped` debug log so skip reasons (e.g. `weekly-not-exhausted`) can be attributed to the evaluated account. - Fixed unawaited promise rejections in JS eval cells crashing the session: a floating rejection now fails the owning cell run (`Unhandled rejection (missing await?): …`) instead of escaping to the global `unhandledRejection` handler, which printed `[Unhandled Rejection]` and killed the process (inline fallback) or tore down the eval worker (dedicated worker). Rejections surfacing after a cell settled are downgraded to a warn log attributed to the finished cell. -### Fixed - - Fixed project `.omp/RULES.md` sticky rules being shadowed by user `~/.omp/agent/RULES.md` rules with the same synthesized `RULES` name, so both user and project sticky rules now inject ([#4739](https://github.com/can1357/oh-my-pi/issues/4739)). -### Fixed - - Fixed bash internal-URL expansion so unresolved literal `memory://` / `skill://` text stays verbatim instead of aborting command execution ([#4737](https://github.com/can1357/oh-my-pi/issues/4737)). - Fixed `agent://` (and the `output()` eval helper) failing with `Not found` for a subagent spawned by another subagent (any spawn chain 2+ levels deep). `artifactsDirsFromRegistry` scanned only each ref's adopted (root-wide) `ArtifactManager` dir, but a subagent's own children are written one level deeper under its `sessionFile`-derived dir — so a live, addressable nested peer's output was unresolvable. The resolver now collects both candidate dirs per registered agent. ([#4650](https://github.com/can1357/oh-my-pi/issues/4650)) - Fixed plan mode to document `local://` artifacts as writable session-local planning files and to carry every pre-approval `local://` artifact into the fresh session created by Approve and Execute. - Fixed the browser tool failing to launch Microsoft Edge-only Windows installs with Puppeteer's empty `Code: 0` launch error by keeping Edge's required `--enable-automation` default while preserving Chrome/Chromium stealth launch defaults. +- Fixed bash/tool command environments inheriting Bun-autoloaded launch `.env.local` values, so nested apps can load their own dotenv values without parent deployment variables taking precedence. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) +- Fixed legacy plugin validation for extension graphs that import JSON with `with { type: "json" }`, leaving JSON files on Bun's native loader instead of parsing them as JavaScript ([#4687](https://github.com/can1357/oh-my-pi/issues/4687)). +- Fixed pasted terminal transcripts beginning with a shell prompt (`$ ...`) being mistaken for local Python shortcuts instead of being submitted as normal prompts ([#4678](https://github.com/can1357/oh-my-pi/issues/4678)). +- Fixed wrapped OAuth copy-URL rows corrupting on paste: continuation chunks no longer carry a leading indent, so a multi-row terminal selection reassembles to the exact authorize URL (browsers strip newlines on paste but preserve or percent-encode embedded spaces, which previously corrupted the URL at every chunk boundary). +- Fixed Windows browser-launch failures being unobservable: the opener now uses `%SystemRoot%`-resolved PowerShell `Start-Process` (via `-EncodedCommand`) instead of `rundll32`, which exits 0 unconditionally. Failures ShellExecute itself reports — missing target, no handler executable, access denied — now surface as non-zero exits and are logged; the encoded payload also keeps OAuth query strings (`&`-bearing) opaque to shell metacharacter parsing. +- Fixed system prompt date rendering to use the host local calendar date instead of UTC. +- Fixed `bash` tool `timeout: 0` so it disables the command deadline instead of falling back to the minimum timeout. +- Fixed `read` and `grep` refusing to access filesystem paths whose names end in a selector-shaped suffix (e.g. `test:1-2`, `log:raw`) by preferring a literal match over the trailing `:` peel when the raw path exists on disk ([#4618](https://github.com/can1357/oh-my-pi/issues/4618)). +- Fixed wrapped Edit-diff rows leaking inverse video into the result card's right-edge padding: a row that broke inside an intra-line highlight left inverse active at the row end, so the frame padding after it rendered as a default-foreground block ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) +- Fixed Edit-diff continuation rows escaping into the line-number column when the row's gutter was left-padded (line number narrower than the widest in the diff) or blanked by the gutter dedup (the bare `+` row of a single-line replacement); such rows now wrap behind a continuation gutter, while body lines that merely start with `|` keep wrapping generically ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) +- Fixed live advisors continuing to use a stale `modelRoles.advisor` selection after `/model` changed the advisor model. ([#4612](https://github.com/can1357/oh-my-pi/issues/4612)) +- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry. The resolver also now honours Claude's per-field merge semantic — `skills` adds to the default `skills/` scan; `commands`/`slash-commands` replace the default `commands/` — so plugins like `{"skills":["./extra-skills"]}` no longer lose their default `skills/` folder while `{"commands":["./admin"]}` still replaces `commands/` as documented. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609)) +- Fixed macOS Backspace on empty search not deleting sessions in the `/resume` picker; Fn+Backspace terminals that deliver `\x7f` instead of `\e[3~` now reach the delete confirmation dialog. ([#4580](https://github.com/can1357/oh-my-pi/pull/4580) by [@JagravNaik](https://github.com/JagravNaik)) +- Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal session title text. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) +- Fixed empty session `.jsonl` files accumulating in `~/.omp/agent/sessions//` after a draft-then-clear exit cycle. `SessionManager.saveDraft(text)` materializes the session file so the draft sidecar has a parent; a subsequent `saveDraft("")` unlinked the sidecar but left the metadata-only JSONL behind (title slot + session header + startup selector entries, ~500–750 B), and `#shouldHaveSessionFile()` could no longer prune it once `#fileIsCurrent`/`#forceFileCreation` were latched. `SessionManager.close()` now drops only draft-owned metadata-only sessions with no saved draft sidecar to reattach to, while keeping real conversations, meaningful non-message entries such as handoff custom messages, explicit `ensureOnDisk()` sessions, drafts still pending for `--resume`, and never-materialized sessions untouched ([#4571](https://github.com/can1357/oh-my-pi/issues/4571)). +- Fixed the advisor being disabled for the entire session when the advisor role resolves to a reasoning model that exposes no controllable effort surface (Devin `devin/glm-5-2*`: `reasoning: true`, `thinking: undefined` — Cascade routes by sibling model id rather than a wire param). `#resolveAdvisorRuntimeDescriptors` in `packages/coding-agent/src/session/agent-session.ts` used to hardcode `ThinkingLevel.Medium`, which tripped `requireSupportedEffort` on the first advisor prompt with `Thinking effort medium is not supported by devin/glm-5-2. Supported efforts:` (empty list). The advisor descriptor now clamps the requested effort against the resolved model via `resolveThinkingLevelForModel` and forwards no explicit effort when the model has no controllable efforts — matching the `auto`-path fix (`clampAutoThinkingEffort`) and the Autonomous Memory stage fix (`clampThinkingLevelForModel`). Explicit `:off` still disables reasoning, and models that support `medium` (e.g. Anthropic) keep receiving it ([#4579](https://github.com/can1357/oh-my-pi/issues/4579)). +- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports `calculateCost` (or `modelsAreEqual` / `getBundledProviders`) from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)). +- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports relocated catalog symbols such as `calculateCost`, `modelsAreEqual`, `getBundledProviders`, `getBundledModel`, or `getBundledModels` from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)). +- Fixed legacy pi extension imports of `DefaultResourceLoader` from `@mariozechner/pi-coding-agent` / `@earendil-works/pi-coding-agent` by adding a compatibility loader shim that translates `resourceLoader` into OMP's native session discovery options. ([#4567](https://github.com/can1357/oh-my-pi/issues/4567)) +- Fixed legacy Pi extension reloads on POSIX so `loadLegacyPiModule` imports the entry through a cache-busting filesystem path, refreshes load-time graph hooks when reloads add new modules, and threads the current load's `?mtime` tag through the extension source graph — relative `./helper.ts` siblings, `#alias/*` package-imports, extension-local bare dependency entries, and their relative children all rekey per reload, so same-process re-imports pick up edits across the whole graph. ([#4565](https://github.com/can1357/oh-my-pi/issues/4565)) +- Fixed bash tool pipeline execution preserving stale upstream output when the final stage was a stripped `head`/`tail` limiter; the tool now runs the command as written so `seq 1 5 | head -n2` returns only `1` and `2`. ([#4562](https://github.com/can1357/oh-my-pi/issues/4562)) +- Fixed the status-line token-rate segment rendering as `/s`, which Ghostty auto-detected as a hyperlink on Ctrl+hover. ([#4541](https://github.com/can1357/oh-my-pi/issues/4541)) +- Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533)) +- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). +- Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.3.11] - 2026-07-06 @@ -36,12 +60,6 @@ - Fixed session titles occasionally showing raw `{"title": "..."}` JSON. Online title generation now always uses the `...` marker prompt instead of a forced `set_title` tool call — hosts that ignored or rejected forced `tool_choice` echoed the prompt's JSON example verbatim as the title — and JSON-shaped responses (bare, code-fenced, marker-wrapped, or truncated) are unwrapped to the bare title. - Fixed Linux startup prompt construction to read the CPU model from `/proc/cpuinfo` instead of `os.cpus()`, avoiding per-core sysfs frequency probes on many-core hosts ([#4712](https://github.com/can1357/oh-my-pi/issues/4712)). - Fixed llama.cpp model discovery to honor per-model `architecture.input_modalities` from `/v1/models`, so router presets that advertise image input are no longer treated as text-only ([#4719](https://github.com/can1357/oh-my-pi/issues/4719)). -### Fixed - -- Fixed bash/tool command environments inheriting Bun-autoloaded launch `.env.local` values, so nested apps can load their own dotenv values without parent deployment variables taking precedence. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) -### Fixed - -- Fixed legacy plugin validation for extension graphs that import JSON with `with { type: "json" }`, leaving JSON files on Bun's native loader instead of parsing them as JavaScript ([#4687](https://github.com/can1357/oh-my-pi/issues/4687)). ## [16.3.10] - 2026-07-06 @@ -61,9 +79,6 @@ - Fixed startup of cached llama.cpp vision models so the initial default/restored model refreshes `/props` metadata before the session exposes it as text-only. - Fixed IRC-woken yielded subagents skipping empty-stop retry because stale yield-termination state carried into the wake turn ([#4658](https://github.com/can1357/oh-my-pi/issues/4658)). - Fixed `irc wait` skipping replies that arrived between wait calls by draining pending IRC asides before honoring queued-interrupt aborts ([#4657](https://github.com/can1357/oh-my-pi/issues/4657)). -### Fixed - -- Fixed pasted terminal transcripts beginning with a shell prompt (`$ ...`) being mistaken for local Python shortcuts instead of being submitted as normal prompts ([#4678](https://github.com/can1357/oh-my-pi/issues/4678)). ## [16.3.9] - 2026-07-06 @@ -82,31 +97,12 @@ - Fixed an issue where local llama.cpp vision models remained text-only after a model refresh, ensuring they are correctly recognized as image-capable when configured as the default or vision role. - Fixed `omp commit` split plans aborting when lock files (such as `bun.lockb`) were staged alongside their manifests by correctly pairing lock files with their corresponding commit groups and properly handling binary files during split execution. - Fixed skill loading to ensure that disabling a higher-priority provider does not drop same-named skills from enabled lower-priority providers. -### Fixed - -- Fixed wrapped OAuth copy-URL rows corrupting on paste: continuation chunks no longer carry a leading indent, so a multi-row terminal selection reassembles to the exact authorize URL (browsers strip newlines on paste but preserve or percent-encode embedded spaces, which previously corrupted the URL at every chunk boundary). -### Fixed - -- Fixed Windows browser-launch failures being unobservable: the opener now uses `%SystemRoot%`-resolved PowerShell `Start-Process` (via `-EncodedCommand`) instead of `rundll32`, which exits 0 unconditionally. Failures ShellExecute itself reports — missing target, no handler executable, access denied — now surface as non-zero exits and are logged; the encoded payload also keeps OAuth query strings (`&`-bearing) opaque to shell metacharacter parsing. -### Fixed - -- Fixed system prompt date rendering to use the host local calendar date instead of UTC. -### Fixed - -- Fixed `bash` tool `timeout: 0` so it disables the command deadline instead of falling back to the minimum timeout. ## [16.3.8] - 2026-07-05 ### Fixed - Fixed the browser tool silently launching without its Puppeteer stealth patch: the patch was keyed to `puppeteer-core@25.1.0` while the resolved dependency had drifted to `25.3.0`, so Bun skipped it and Chrome ran with the `Runtime.enable` automation tell re-enabled. Regenerated the stealth patch against 25.3.0 and pinned `puppeteer-core` so the exact-version patch cannot silently deactivate on future drift. -### Fixed - -- Fixed `read` and `grep` refusing to access filesystem paths whose names end in a selector-shaped suffix (e.g. `test:1-2`, `log:raw`) by preferring a literal match over the trailing `:` peel when the raw path exists on disk ([#4618](https://github.com/can1357/oh-my-pi/issues/4618)). -### Fixed - -- Fixed wrapped Edit-diff rows leaking inverse video into the result card's right-edge padding: a row that broke inside an intra-line highlight left inverse active at the row end, so the frame padding after it rendered as a default-foreground block ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) -- Fixed Edit-diff continuation rows escaping into the line-number column when the row's gutter was left-padded (line number narrower than the widest in the diff) or blanked by the gutter dedup (the bare `+` row of a single-line replacement); such rows now wrap behind a continuation gutter, while body lines that merely start with `|` keep wrapping generically ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) ## [16.3.7] - 2026-07-05 @@ -131,7 +127,6 @@ - Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal text. - Fixed empty session `.jsonl` files accumulating in the sessions directory after a draft-then-clear exit cycle. - Fixed advisor being disabled for the entire session when resolving to a reasoning model with no controllable effort surface (e.g., `devin/glm-5-2*`). -- Fixed live advisors continuing to use a stale `modelRoles.advisor` selection after `/model` changed the advisor model. ([#4612](https://github.com/can1357/oh-my-pi/issues/4612)) - Fixed legacy extension plugin validation failures by re-exporting relocated catalog symbols (such as `calculateCost`, `modelsAreEqual`, and `getBundledProviders`) through the legacy `pi-ai` root shim. - Fixed legacy Pi extension imports of `DefaultResourceLoader` by adding a compatibility loader shim that translates `resourceLoader` into native session discovery options. - Fixed legacy Pi extension reloads on POSIX to ensure same-process re-imports pick up edits across the entire dependency graph. @@ -168,20 +163,6 @@ - Aborted underlying MCP calls when proxy tool timeouts fire. - Surfaced unexpected JS eval worker exits via close listeners to prevent silent hangs. - Cached failed `!command` config resolutions and timed out extension dynamic model fetches after 15 seconds. -- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry. The resolver also now honours Claude's per-field merge semantic — `skills` adds to the default `skills/` scan; `commands`/`slash-commands` replace the default `commands/` — so plugins like `{"skills":["./extra-skills"]}` no longer lose their default `skills/` folder while `{"commands":["./admin"]}` still replaces `commands/` as documented. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609)) -- Fixed macOS Backspace on empty search not deleting sessions in the `/resume` picker; Fn+Backspace terminals that deliver `\x7f` instead of `\e[3~` now reach the delete confirmation dialog. ([#4580](https://github.com/can1357/oh-my-pi/pull/4580) by [@JagravNaik](https://github.com/JagravNaik)) -- Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal session title text. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) -- Fixed empty session `.jsonl` files accumulating in `~/.omp/agent/sessions//` after a draft-then-clear exit cycle. `SessionManager.saveDraft(text)` materializes the session file so the draft sidecar has a parent; a subsequent `saveDraft("")` unlinked the sidecar but left the metadata-only JSONL behind (title slot + session header + startup selector entries, ~500–750 B), and `#shouldHaveSessionFile()` could no longer prune it once `#fileIsCurrent`/`#forceFileCreation` were latched. `SessionManager.close()` now drops only draft-owned metadata-only sessions with no saved draft sidecar to reattach to, while keeping real conversations, meaningful non-message entries such as handoff custom messages, explicit `ensureOnDisk()` sessions, drafts still pending for `--resume`, and never-materialized sessions untouched ([#4571](https://github.com/can1357/oh-my-pi/issues/4571)). -- Fixed the advisor being disabled for the entire session when the advisor role resolves to a reasoning model that exposes no controllable effort surface (Devin `devin/glm-5-2*`: `reasoning: true`, `thinking: undefined` — Cascade routes by sibling model id rather than a wire param). `#resolveAdvisorRuntimeDescriptors` in `packages/coding-agent/src/session/agent-session.ts` used to hardcode `ThinkingLevel.Medium`, which tripped `requireSupportedEffort` on the first advisor prompt with `Thinking effort medium is not supported by devin/glm-5-2. Supported efforts:` (empty list). The advisor descriptor now clamps the requested effort against the resolved model via `resolveThinkingLevelForModel` and forwards no explicit effort when the model has no controllable efforts — matching the `auto`-path fix (`clampAutoThinkingEffort`) and the Autonomous Memory stage fix (`clampThinkingLevelForModel`). Explicit `:off` still disables reasoning, and models that support `medium` (e.g. Anthropic) keep receiving it ([#4579](https://github.com/can1357/oh-my-pi/issues/4579)). -- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports `calculateCost` (or `modelsAreEqual` / `getBundledProviders`) from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)). -- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports relocated catalog symbols such as `calculateCost`, `modelsAreEqual`, `getBundledProviders`, `getBundledModel`, or `getBundledModels` from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)). -- Fixed legacy pi extension imports of `DefaultResourceLoader` from `@mariozechner/pi-coding-agent` / `@earendil-works/pi-coding-agent` by adding a compatibility loader shim that translates `resourceLoader` into OMP's native session discovery options. ([#4567](https://github.com/can1357/oh-my-pi/issues/4567)) -- Fixed legacy Pi extension reloads on POSIX so `loadLegacyPiModule` imports the entry through a cache-busting filesystem path, refreshes load-time graph hooks when reloads add new modules, and threads the current load's `?mtime` tag through the extension source graph — relative `./helper.ts` siblings, `#alias/*` package-imports, extension-local bare dependency entries, and their relative children all rekey per reload, so same-process re-imports pick up edits across the whole graph. ([#4565](https://github.com/can1357/oh-my-pi/issues/4565)) -- Fixed bash tool pipeline execution preserving stale upstream output when the final stage was a stripped `head`/`tail` limiter; the tool now runs the command as written so `seq 1 5 | head -n2` returns only `1` and `2`. ([#4562](https://github.com/can1357/oh-my-pi/issues/4562)) -- Fixed the status-line token-rate segment rendering as `/s`, which Ghostty auto-detected as a hyperlink on Ctrl+hover. ([#4541](https://github.com/can1357/oh-my-pi/issues/4541)) -### Fixed - -- Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533)) ## [16.3.6] - 2026-07-04 @@ -218,9 +199,6 @@ - Fixed ACP `terminal/create` sending the bash tool's full shell line in `command` with no `args`, which broke spec-conformant clients that spawn `command`+`args` directly (no implicit shell) — any command containing a space, pipe, `&&`, redirect, or `$(...)` failed with `ENOENT` and the agent silently degraded to read-only tools. The bash tool now wraps the shell line before calling `clientBridge.createTerminal`, reusing the same shell binary + args the local `bash-executor` resolves via `settings.getShellConfig()` (Git Bash / `bash.exe` on Windows, `$SHELL` with `sh` fallback on POSIX) so bash semantics — `$VAR`, `$(...)`, `source`, POSIX quoting, `-l` — are preserved on both platforms. ([#4333](https://github.com/can1357/oh-my-pi/issues/4333)) - Fixed inference worker subprocesses (TTS, STT, tiny-model, mnemopi embeddings) discarding stderr, which left every unexpected exit — most visibly the local Kokoro TTS worker's recurring `exit code 7` crash loop — undiagnosable from the parent's logs. `createWorkerSubprocess` now pipes stderr without starting a live read while the worker is idle, then drains the stream after `onExit`, emits captured lines to `logger.debug` under an ` stderr` message, and keeps the last 16 KiB in a bounded ring that gets appended to the `Error` surfaced through `onError`. The exit surface is synchronized with the post-exit drain via `SpawnedSubprocess.stderrDrained`, so the full native trace shows up on the `tts: worker error` line without reintroducing event-loop liveness from unref'd workers. ([#4324](https://github.com/can1357/oh-my-pi/issues/4324)) - Fixed Windows session tail loss after atomic compaction rewrites by fencing append writers during full-file replacement and gating the atomic publish on a `commitGuard` that the storage backend checks synchronously before rename, so a concurrent `flushSync` (Ctrl+C / session-exit) is not overwritten by the stale body serialized before it ran. Covers post-compaction prompts, tool results, title changes, and exit diagnostics on the current JSONL path ([#4338](https://github.com/can1357/oh-my-pi/issues/4338)). -### Fixed - -- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). ## [16.3.4] - 2026-07-03 @@ -918,9 +896,6 @@ ### Removed - Removed the `readHashLines` setting (the "Hash Lines" toggle under Files → Reading). Hashline read/search anchors (`[PATH#TAG]` snapshot headers plus `LINE:content`) are now driven solely by `edit.mode === "hashline"`: the toggle was redundant when off (anchors are already suppressed for non-hashline edit modes) and a footgun when on (turning it off left the default hashline edit tool with no addressable anchors, since `read` then skips recording the snapshot tag). Existing configs are migrated automatically by dropping the stale key. -### Added - -- Typing `#` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline. Naming the type (`pr #3164` / `issue #3164`) constrains the candidates to that kind, and embedded hashes like `owner/repo#N`, `foo#N`, or URL fragments are left untouched ([#3218](https://github.com/can1357/oh-my-pi/issues/3218)) ## [16.1.12] - 2026-06-21 @@ -1450,10 +1425,6 @@ - Fixed task subagents to install their configured ordered model candidates as child-session retry fallback chains, so retryable provider failures can advance to the next subagent model instead of failing the worker ([#2750](https://github.com/can1357/oh-my-pi/issues/2750)). - Fixed empty reasonless aborted assistant turns to auto-retry without switching model fallback, so transient provider-side aborts after tool results do not end headless sessions ([#2685](https://github.com/can1357/oh-my-pi/issues/2685)). - -### Fixed - -- Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.0.1] - 2026-06-15 ### Breaking Changes diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index b2ab461d3..adc4c7148 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -5,6 +5,10 @@ ### Fixed - Fixed selector rendering when a legacy theme omits symbol settings by falling back to an ASCII cursor instead of crashing ([#4745](https://github.com/can1357/oh-my-pi/issues/4745)). +- Kept slash command autocomplete rows compact by truncating descriptions instead of wrapping them into multi-line blocks. +- Fixed mid-prompt skill autocomplete so Tab and Enter accept the highlighted `/skill:` suggestion and Backspace dismisses the popup immediately after removing the triggering slash ([#4619](https://github.com/can1357/oh-my-pi/issues/4619)). +- Fixed submitted slash-command arguments treating `@` file-reference tokens as prompt-composer autocomplete triggers when the command does not define argument completions. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) +- Fixed box-drawing tree lines (`├── item` — directory layouts, decision trees) in prose shearing apart when they wrap: continuation rows now hang under the node text with ancestor rails carried through (`├` → `│`, `└` → blank) instead of restarting at column 0. Applies to prose paragraphs (including inside blockquotes) only when a line with a branch-connector prefix (`├──`, `└─`, …) actually overflows; fitting lines, non-tree prose, and code blocks render byte-for-byte as before. ## [16.3.10] - 2026-07-06 @@ -18,8 +22,6 @@ - Fixed autocompletion for absolute paths (such as `/tmp/...` or `/Users/...`) at the start of a prompt, ensuring they fall back to file-path completion instead of being incorrectly treated as slash commands. - Updated absolute path autocompletion behavior so that accepting a suggestion inserts the path without submitting the prompt. -- Kept slash command autocomplete rows compact by truncating descriptions instead of wrapping them into multi-line blocks. -- Fixed mid-prompt skill autocomplete so Tab and Enter accept the highlighted `/skill:` suggestion and Backspace dismisses the popup immediately after removing the triggering slash ([#4619](https://github.com/can1357/oh-my-pi/issues/4619)). ## [16.3.7] - 2026-07-05 @@ -27,8 +29,6 @@ - Fixed an issue where `@` file-reference tokens in slash-command arguments incorrectly triggered prompt-composer autocompletion when the command did not define argument completions. - Fixed a memory leak caused by unbounded map growth in the image budget cache. -- Fixed submitted slash-command arguments treating `@` file-reference tokens as prompt-composer autocomplete triggers when the command does not define argument completions. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) -- Fixed box-drawing tree lines (`├── item` — directory layouts, decision trees) in prose shearing apart when they wrap: continuation rows now hang under the node text with ancestor rails carried through (`├` → `│`, `└` → blank) instead of restarting at column 0. Applies to prose paragraphs (including inside blockquotes) only when a line with a branch-connector prefix (`├──`, `└─`, …) actually overflows; fitting lines, non-tree prose, and code blocks render byte-for-byte as before. ## [16.3.6] - 2026-07-04 diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index f5ebdd7a3..a6f89cf70 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -5,6 +5,7 @@ ### Added - Added `postmortem.interceptUnhandledRejections()` to register interceptors consulted before an unhandled rejection tears the process down; a consuming interceptor (e.g. the JS eval runtime claiming rejections floated by user cell code) keeps the process alive and owns reporting. + ### Fixed - Fixed child shell environment filtering to drop launch-directory `.env.local` values that Bun auto-loaded before OMP starts command shells. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) From 53df3c82b7f639b7937979a49cc66188927c4f37 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:37:43 +0200 Subject: [PATCH 90/91] style: applied biome formatting to merged sources --- packages/agent/src/agent-loop.ts | 4 ++-- packages/ai/src/providers/openai-codex-responses.ts | 6 +++++- packages/coding-agent/src/session/agent-session.ts | 10 ++++++++-- packages/coding-agent/src/tools/bash.ts | 5 ++++- .../coding-agent/test/agent-session-concurrent.test.ts | 1 - .../test/agent-session-magic-keywords.test.ts | 2 +- packages/coding-agent/test/modes/workflow.test.ts | 7 ++++++- packages/tui/src/components/editor.ts | 3 ++- 8 files changed, 28 insertions(+), 10 deletions(-) diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index a69fdf2e3..c8b1b82e5 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -944,7 +944,8 @@ async function runLoopBody( // only the matching call is blamed and siblings stay neutral. const scopedAbort = toolScopedAbortReason(signal); const toolCallAbortMessages = - message.toolCallAbortMessages ?? (scopedAbort ? buildToolCallAbortMessages(message, scopedAbort) : undefined); + message.toolCallAbortMessages ?? + (scopedAbort ? buildToolCallAbortMessages(message, scopedAbort) : undefined); const toolResults: ToolResultMessage[] = []; for (const toolCall of toolCalls) { const errorMessage = toolCallAbortMessages?.[toolCall.id] ?? message.errorMessage; @@ -1683,7 +1684,6 @@ export function abortReasonText(signal: AbortSignal | undefined): string { return "Request was aborted"; } - function emitAbortedAssistantMessage( partialMessage: AssistantMessage | null, addedPartial: boolean, diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 5030cc67c..80baabe8d 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -1242,7 +1242,11 @@ const CODEX_STALE_PREVIOUS_RESPONSE_CODES: Record = { function isCodexStalePreviousResponseError(error: unknown): boolean { if (!(error instanceof Error)) return false; - if ("code" in error && typeof error.code === "string" && Object.hasOwn(CODEX_STALE_PREVIOUS_RESPONSE_CODES, error.code)) { + if ( + "code" in error && + typeof error.code === "string" && + Object.hasOwn(CODEX_STALE_PREVIOUS_RESPONSE_CODES, error.code) + ) { return true; } // Message-based fallback for providers/proxies that report the condition diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 63d03baec..85e544ca1 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2415,7 +2415,9 @@ export class AgentSession { #advisorRuntimeSignature(config: AdvisorConfig, slug: string, model: Model, thinkingLevel: ThinkingLevel): string { const tools = config.tools?.length ? config.tools.join("\u001e") : ""; const instructions = config.instructions?.trim() ?? ""; - return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join("\u001f"); + return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join( + "\u001f", + ); } #advisorRuntimeMatchesCurrentConfig(): boolean { @@ -7397,7 +7399,11 @@ export class AgentSession { timestamp, }); } - if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text) && this.getActiveToolNames().includes("task")) { + if ( + this.#magicKeywordEnabled("workflow") && + containsWorkflow(text) && + this.getActiveToolNames().includes("task") + ) { keywordNotices.push({ role: "custom", customType: "workflow-notice", diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index 56f5bcaee..14e67375b 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -389,7 +389,10 @@ export class BashTool implements AgentTool { expect(toolResultText(matchedToolCallContent.id)).toBe(""); }); - it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; diff --git a/packages/coding-agent/test/agent-session-magic-keywords.test.ts b/packages/coding-agent/test/agent-session-magic-keywords.test.ts index d675005ed..3a10dfe8b 100644 --- a/packages/coding-agent/test/agent-session-magic-keywords.test.ts +++ b/packages/coding-agent/test/agent-session-magic-keywords.test.ts @@ -12,8 +12,8 @@ import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking"; -import { type } from "arktype"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; const mockTaskTool: AgentTool = { name: "task", diff --git a/packages/coding-agent/test/modes/workflow.test.ts b/packages/coding-agent/test/modes/workflow.test.ts index 32b01f31b..61a5ae838 100644 --- a/packages/coding-agent/test/modes/workflow.test.ts +++ b/packages/coding-agent/test/modes/workflow.test.ts @@ -1,6 +1,11 @@ import { beforeAll, describe, expect, it } from "bun:test"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { containsWorkflow, highlightWorkflow, renderWorkflowNotice, WORKFLOW_NOTICE } from "@oh-my-pi/pi-coding-agent/modes/workflow"; +import { + containsWorkflow, + highlightWorkflow, + renderWorkflowNotice, + WORKFLOW_NOTICE, +} from "@oh-my-pi/pi-coding-agent/modes/workflow"; beforeAll(() => { // highlightWorkflow reads the global theme's color mode. diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index bcaba89a8..5d44e03c9 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -2916,7 +2916,8 @@ export class Editor implements Component, Focusable { // rewriting the user's `/tmp` to `/skill:…`. const lowerToken = token.slice(1).toLowerCase(); if (scoreCommandTextMatch(lowerToken, item.value.toLowerCase()) > 0) return true; - if (item.description && scoreCommandTextMatch(lowerToken, item.description.toLowerCase()) > 0) return true; + if (item.description && scoreCommandTextMatch(lowerToken, item.description.toLowerCase()) > 0) + return true; } } return false; From 68da3dca8a7c55c655587dfa1acfce32a737ece9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 8 Jul 2026 15:43:56 +0200 Subject: [PATCH 91/91] test: aligned merged regression tests with current main contracts - handoff veto mock returns a promise and threshold expectation includes onSwitchCancelled - codex websocket stale-anchor stats assertion tolerates lastTurn debug payload --- packages/ai/test/openai-codex-stream.test.ts | 2 +- .../coding-agent/test/agent-session-handoff.test.ts | 11 ++++------- 2 files changed, 5 insertions(+), 8 deletions(-) diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 5c0425de8..e7fa92c2e 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -2808,7 +2808,7 @@ describe("openai-codex streaming", () => { sessionId: "ws-proxy-stale-anchor-session", providerSessionState, }); - expect(stats).toEqual({ + expect(stats).toMatchObject({ fullContextRequests: 2, deltaRequests: 1, lastInputItems: (retryInput as unknown[]).length, diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index 8b00a4e1c..47c30fe9c 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -1420,6 +1420,7 @@ describe("AgentSession handoff", () => { expect(handoffSpy).toHaveBeenCalledWith(expect.stringContaining("Threshold-triggered maintenance"), { autoTriggered: true, signal: expect.anything(), + onSwitchCancelled: expect.any(Function), }); expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(1); const endEvents = events.filter(event => event.type === "auto_compaction_end"); @@ -1701,13 +1702,9 @@ describe("AgentSession handoff", () => { modelRegistry, ); vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch"); - const emit = extensionRunner.emit.bind(extensionRunner); - const emitSpy = vi.spyOn(extensionRunner, "emit").mockImplementation(event => { - if (event.type === "session_before_switch") { - return { cancel: true }; - } - return emit(event); - }); + const emitSpy = vi.spyOn(extensionRunner, "emit").mockImplementation((async () => ({ + cancel: true, + })) as ExtensionRunner["emit"]); await session.dispose(); session = new AgentSession({