From 38a5c8a89335cdcaacf4289dacd834dd75194e5d Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Fri, 3 Jul 2026 07:57:19 +0900 Subject: [PATCH 001/205] feat(ask): add rich interactive dialog Adds the rich TUI ask dialog, additive schema fields, ask.enabled tool gating, note/preview/header support, chat redirect, and timeout behavior that defers while nested prompts are active instead of discarding user input. Op: extend --- .../src/config/settings-schema.ts | 11 + .../src/extensibility/extensions/types.ts | 37 + .../src/modes/components/ask-dialog.ts | 837 ++++++++++++++++++ .../controllers/extension-ui-controller.ts | 232 ++++- packages/coding-agent/src/tools/ask.ts | 244 ++++- packages/coding-agent/src/tools/index.ts | 1 + .../test/modes/components/ask-dialog.test.ts | 738 +++++++++++++++ .../modes/components/settings-layout.test.ts | 10 + packages/coding-agent/test/tools/ask.test.ts | 148 +++- .../coding-agent/test/tools/index.test.ts | 21 + 10 files changed, 2231 insertions(+), 48 deletions(-) create mode 100644 packages/coding-agent/src/modes/components/ask-dialog.ts create mode 100644 packages/coding-agent/test/modes/components/ask-dialog.test.ts diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 2316034da..96bfcb058 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -3630,6 +3630,17 @@ export const SETTINGS_SCHEMA = { }, }, + "ask.enabled": { + type: "boolean", + default: true, + ui: { + tab: "tools", + group: "Available Tools", + label: "Ask", + description: "Enable the ask tool for interactive user questions", + }, + }, + "browser.enabled": { type: "boolean", default: true, diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index b93535142..03e3da10e 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -113,6 +113,37 @@ export interface ExtensionUISelectOption { export type ExtensionUISelectItem = string | ExtensionUISelectOption; +export interface ExtensionAskDialogOption { + label: string; + description?: string; + preview?: string; +} + +export interface ExtensionAskDialogQuestion { + id: string; + question: string; + header?: string; + options: ExtensionAskDialogOption[]; + multi?: boolean; + recommended?: number; +} + +export interface ExtensionAskDialogResultItem { + id: string; + question: string; + options: string[]; + multi: boolean; + selectedOptions: string[]; + customInput?: string; + note?: string; + timedOut?: boolean; +} + +export interface ExtensionAskDialogResult { + kind: "submit"; + results: ExtensionAskDialogResultItem[]; +} + export function getExtensionUISelectOptionLabel(option: ExtensionUISelectItem): string { return typeof option === "string" ? option : option.label; } @@ -186,6 +217,12 @@ export interface ExtensionUIContext { /** Show a text input dialog. */ input(title: string, placeholder?: string, dialogOptions?: ExtensionUIDialogOptions): Promise; + /** Show the rich ask dialog when the interactive TUI surface is available. */ + askDialog?( + questions: ExtensionAskDialogQuestion[], + dialogOptions?: ExtensionUIDialogOptions, + ): Promise; + /** Show a notification to the user. */ notify(message: string, type?: "info" | "warning" | "error"): void; diff --git a/packages/coding-agent/src/modes/components/ask-dialog.ts b/packages/coding-agent/src/modes/components/ask-dialog.ts new file mode 100644 index 000000000..de96dcef3 --- /dev/null +++ b/packages/coding-agent/src/modes/components/ask-dialog.ts @@ -0,0 +1,837 @@ +import { + type Component, + Ellipsis, + Markdown, + type MarkdownTheme, + matchesKey, + padding, + renderInlineMarkdown, + replaceTabs, + ScrollView, + type Tab, + TabBar, + Text, + type TUI, + truncateToWidth, + visibleWidth, + wrapTextWithAnsi, +} from "@oh-my-pi/pi-tui"; +import type { + ExtensionAskDialogQuestion, + ExtensionAskDialogResult, + ExtensionAskDialogResultItem, +} from "../../extensibility/extensions"; +import { getTabBarTheme } from "../shared"; +import { getMarkdownTheme, highlightCode, theme } from "../theme/theme"; +import { matchesSelectCancel, matchesSelectDown, matchesSelectUp } from "../utils/keybinding-matchers"; +import { CountdownTimer } from "./countdown-timer"; +import { bottomBorder, divider, row, topBorder } from "./overlay-box"; +import { handleTabSwitchKey } from "./selector-helpers"; + +const OTHER_OPTION = "Other (type your own)"; +const CHAT_ABOUT_THIS_OPTION = "Chat about this"; +const NEXT_OPTION = "Next →"; +const SUBMIT_OPTION = "Submit"; + +const MIN_BODY_ROWS = 5; +const PREVIEW_MIN_WIDTH = 40; +const SIDE_BY_SIDE_LIST_MIN_WIDTH = 30; +const SIDE_BY_SIDE_GAP_WIDTH = 3; +const MAX_HEADER_CHIP_WIDTH = 16; +const PREVIEW_HEADER = "Preview"; + +interface AskDialogCallbacks { + onSubmit(result: ExtensionAskDialogResult): void; + onCancel(): void; + onChat(): void; + onPrompt(title: string, prefill?: string): Promise; +} + +interface AskDialogOptions { + timeout?: number; + onTimeout?: () => void; + tui?: TUI; +} + +interface QuestionState { + selectedOptions: Set; + customInput: string | undefined; + note: string | undefined; + noteRowKey: string | undefined; + cursorIndex: number; + scrollOffset: number; + timedOut: boolean; +} + +type QuestionRowKind = "option" | "other" | "next" | "chat"; +type SubmitRowKind = "submit" | "chat"; + +interface QuestionRow { + kind: QuestionRowKind; + key: string; + label: string; + optionIndex: number | undefined; +} + +interface SubmitRow { + kind: SubmitRowKind; + key: string; + label: string; +} + +interface RenderedList { + lines: string[]; + scrollOffset: number; + indicator: string; +} + +interface PreviewSegment { + kind: "markdown" | "code"; + text: string; + language: string | undefined; +} + +function clamp(value: number, min: number, max: number): number { + return Math.max(min, Math.min(value, max)); +} + +function stripRecommendedSuffix(label: string): string { + const suffix = " (Recommended)"; + return label.endsWith(suffix) ? label.slice(0, -suffix.length) : label; +} + +function questionTabLabel(question: ExtensionAskDialogQuestion, index: number): string { + const base = question.header?.trim() || question.id || `Q${index + 1}`; + return truncateToWidth(replaceTabs(base), MAX_HEADER_CHIP_WIDTH, Ellipsis.Unicode); +} + +function renderQuestionTitle(question: ExtensionAskDialogQuestion, index: number, width: number): string[] { + const chip = question.header?.trim() ? theme.fg("accent", `[${questionTabLabel(question, index)}] `) : ""; + const mdTheme = getMarkdownTheme(); + const questionText = renderInlineMarkdown(replaceTabs(question.question), mdTheme, t => theme.fg("text", t)); + const titleWidth = Math.max(1, width - visibleWidth(chip)); + const wrapped = wrapTextWithAnsi(questionText, titleWidth); + if (wrapped.length === 0) return [chip.trimEnd()]; + return wrapped.map((line, lineIndex) => + lineIndex === 0 ? `${chip}${line}` : `${padding(visibleWidth(chip))}${line}`, + ); +} + +function splitPreviewSegments(preview: string): PreviewSegment[] { + const segments: PreviewSegment[] = []; + const markdownBuffer: string[] = []; + let fenceChar: string | undefined; + let fenceLength = 0; + let fenceLanguage: string | undefined; + let codeBuffer: string[] = []; + + const flushMarkdown = (): void => { + if (markdownBuffer.length === 0) return; + segments.push({ kind: "markdown", text: markdownBuffer.join("\n"), language: undefined }); + markdownBuffer.length = 0; + }; + const flushCode = (): void => { + segments.push({ kind: "code", text: codeBuffer.join("\n"), language: fenceLanguage }); + codeBuffer = []; + fenceChar = undefined; + fenceLength = 0; + fenceLanguage = undefined; + }; + + for (const line of replaceTabs(preview).split("\n")) { + const fenceMatch = /^(\s{0,3})(`{3,}|~{3,})(.*)$/.exec(line); + if (fenceChar !== undefined) { + if (fenceMatch) { + const marker = fenceMatch[2] ?? ""; + const info = fenceMatch[3]?.trim() ?? ""; + if (marker.startsWith(fenceChar) && marker.length >= fenceLength && info === "") { + flushCode(); + continue; + } + } + codeBuffer.push(line); + continue; + } + if (fenceMatch) { + flushMarkdown(); + const marker = fenceMatch[2] ?? ""; + fenceChar = marker[0]; + fenceLength = marker.length; + fenceLanguage = fenceMatch[3]?.trim().split(/\s+/, 1)[0] || undefined; + codeBuffer = []; + continue; + } + markdownBuffer.push(line); + } + + if (fenceChar !== undefined) { + segments.push({ kind: "code", text: codeBuffer.join("\n"), language: fenceLanguage }); + } else { + flushMarkdown(); + } + return segments; +} + +function renderPreviewContent(preview: string | undefined, width: number): string[] { + if (!preview?.trim()) return [theme.fg("muted", "No preview for this option.")]; + const out: string[] = []; + const mdTheme = getMarkdownTheme(); + const accentStyle = { color: (text: string) => theme.fg("muted", text) }; + for (const segment of splitPreviewSegments(preview)) { + if (segment.kind === "code") { + const highlighted = highlightCode(segment.text, segment.language); + const text = new Text(highlighted.join("\n"), 0, 0); + out.push(...text.render(Math.max(1, width))); + continue; + } + const markdown = new Markdown(segment.text, 0, 0, mdTheme, accentStyle); + out.push(...markdown.render(Math.max(1, width))); + } + return out.length > 0 ? out : [theme.fg("muted", "No preview for this option.")]; +} + +function normalizedInlineInput(input: string): string { + return replaceTabs(input).replace(/\s+/g, " ").trim(); +} + +function renderAnswerSummary(question: ExtensionAskDialogQuestion, state: QuestionState): string { + const selected = question.options.map(option => option.label).filter(label => state.selectedOptions.has(label)); + if (question.multi) { + const answers = [...selected]; + if (state.customInput !== undefined) answers.push(`Other: “${normalizedInlineInput(state.customInput)}”`); + return answers.length > 0 ? answers.join(", ") : theme.fg("warning", "unanswered"); + } + if (state.customInput !== undefined) return `“${normalizedInlineInput(state.customInput)}”`; + if (selected.length === 0) return theme.fg("warning", "unanswered"); + return selected[0] ?? theme.fg("warning", "unanswered"); +} + +function clearNote(state: QuestionState): void { + state.note = undefined; + state.noteRowKey = undefined; +} + +function clearNoteIfRow(state: QuestionState, rowKey: string): void { + if (state.noteRowKey === rowKey) clearNote(state); +} + +function clearNoteUnlessRow(state: QuestionState, rowKey: string): void { + if (state.noteRowKey !== undefined && state.noteRowKey !== rowKey) clearNote(state); +} + +function noteForSubmittedAnswer(question: ExtensionAskDialogQuestion, state: QuestionState): string | undefined { + if (state.note === undefined || state.noteRowKey === undefined) return undefined; + if (state.noteRowKey === "other") return state.customInput !== undefined ? state.note : undefined; + const match = /^option:(\d+)$/.exec(state.noteRowKey); + const optionIndex = match?.[1] === undefined ? Number.NaN : Number.parseInt(match[1], 10); + const option = Number.isInteger(optionIndex) ? question.options[optionIndex] : undefined; + return option && state.selectedOptions.has(option.label) ? state.note : undefined; +} + +function optionMarker(question: ExtensionAskDialogQuestion, checked: boolean): string { + if (question.multi) return checked ? theme.checkbox.checked : theme.checkbox.unchecked; + return checked ? theme.radio.selected : theme.radio.unselected; +} + +function renderRowLabel( + rowItem: QuestionRow, + question: ExtensionAskDialogQuestion, + state: QuestionState, + selected: boolean, + mdTheme: MarkdownTheme, + width: number, +): string[] { + const isOption = rowItem.kind === "option"; + const isOther = rowItem.kind === "other"; + const checked = isOption + ? state.selectedOptions.has(stripRecommendedSuffix(rowItem.label)) + : isOther && state.customInput !== undefined; + const color = selected ? "accent" : checked ? "toolOutput" : "text"; + const marker = + isOption || isOther ? `${theme.fg(checked ? "success" : "dim", optionMarker(question, checked))} ` : " "; + const cursor = selected ? theme.fg("accent", `${theme.nav.cursor} `) : " "; + const label = renderInlineMarkdown(rowItem.label, mdTheme, t => theme.fg(color, t)); + const noteMarker = state.note && state.noteRowKey === rowItem.key ? theme.fg("success", " ✎ note") : ""; + const firstLine = `${cursor}${marker}${label}${noteMarker}`; + const lines = [truncateToWidth(firstLine, width, Ellipsis.Unicode)]; + if (rowItem.kind === "option") { + const option = question.options[rowItem.optionIndex ?? -1]; + if (option?.description?.trim()) { + const description = renderInlineMarkdown(option.description.trim(), mdTheme, t => theme.fg("muted", t)); + const wrapped = wrapTextWithAnsi(description, Math.max(1, width - 6)); + for (const line of wrapped.slice(0, 2)) { + lines.push(` ${truncateToWidth(line, Math.max(1, width - 6), Ellipsis.Unicode)}`); + } + } + } + if (isOther && state.customInput !== undefined) { + const preview = replaceTabs(state.customInput).replace(/\s+/g, " ").trim(); + lines.push(theme.fg("muted", ` ${truncateToWidth(preview, Math.max(1, width - 6), Ellipsis.Unicode)}`)); + } + return lines; +} + +export class AskDialogComponent implements Component { + #states: QuestionState[]; + #activeTabIndex = 0; + #submitCursorIndex = 0; + #submitScrollOffset = 0; + #remainingSeconds: number | undefined; + #countdown: CountdownTimer | undefined; + #promptActive = false; + #timeoutExpired = false; + #closed = false; + #tabBar: TabBar | undefined; + + constructor( + private readonly questions: ExtensionAskDialogQuestion[], + private readonly callbacks: AskDialogCallbacks, + private readonly options: AskDialogOptions = {}, + ) { + this.#states = questions.map(question => { + const recommended = Number.isInteger(question.recommended) ? question.recommended : 0; + const maxIndex = Math.max(0, question.options.length - 1); + return { + selectedOptions: new Set(), + customInput: undefined, + note: undefined, + noteRowKey: undefined, + cursorIndex: clamp(recommended ?? 0, 0, maxIndex), + scrollOffset: 0, + timedOut: false, + }; + }); + if (options.timeout && options.timeout > 0) { + this.#countdown = new CountdownTimer( + options.timeout, + options.tui, + seconds => { + this.#remainingSeconds = seconds; + }, + () => this.#handleTimeout(), + ); + } + } + + invalidate(): void { + this.#tabBar?.invalidate(); + } + + dispose(): void { + this.#closed = true; + this.#countdown?.dispose(); + } + + handleInput(keyData: string): void { + if (this.#closed || this.#promptActive) return; + if (matchesSelectCancel(keyData)) { + this.#finishCancel(); + return; + } + if (this.#hasSubmitTab() && handleTabSwitchKey(keyData, direction => this.#switchTab(direction))) { + this.#requestRender(); + return; + } + if (this.#isSubmitTab()) { + this.#handleSubmitTabInput(keyData); + return; + } + this.#handleQuestionInput(keyData); + } + + render(width: number): readonly string[] { + const height = Math.max(12, process.stdout.rows || 40); + const innerWidth = Math.max(1, width - 4); + const headerLines = this.#renderHeader(innerWidth); + const fixedRows = 1 + headerLines.length + 1 + 1 + 1; + const bodyRows = Math.max(MIN_BODY_ROWS, height - fixedRows); + const bodyLines = this.#isSubmitTab() + ? this.#renderSubmitBody(innerWidth, bodyRows) + : this.#renderQuestionBody(innerWidth, bodyRows); + const footer = this.#footerHintText(bodyLines.indicator); + return [ + topBorder(width, this.#titleText()), + ...headerLines.map(line => row(line, width)), + divider(width), + ...bodyLines.lines.map(line => row(line, width)), + divider(width), + row(theme.fg("dim", footer), width), + bottomBorder(width), + ]; + } + + #titleText(): string { + return this.#remainingSeconds === undefined ? "Ask" : `Ask (${this.#remainingSeconds}s)`; + } + + #hasSubmitTab(): boolean { + return this.questions.length > 1; + } + + #submitTabIndex(): number { + return this.questions.length; + } + + #isSubmitTab(): boolean { + return this.#hasSubmitTab() && this.#activeTabIndex === this.#submitTabIndex(); + } + + #currentQuestionIndex(): number { + return clamp(this.#activeTabIndex, 0, Math.max(0, this.questions.length - 1)); + } + + #requestRender(): void { + this.options.tui?.requestRender(); + } + + #renderHeader(width: number): string[] { + const lines: string[] = []; + if (this.#hasSubmitTab()) { + const tabs: Tab[] = [ + ...this.questions.map((question, index) => ({ + id: String(index), + label: questionTabLabel(question, index), + })), + { id: "submit", label: "Submit" }, + ]; + this.#tabBar = new TabBar("", tabs, getTabBarTheme(), this.#activeTabIndex); + this.#tabBar.showHint = false; + lines.push(...this.#tabBar.render(width)); + } + if (this.#isSubmitTab()) { + lines.push(theme.bold(theme.fg("accent", "Review answers"))); + return lines; + } + const questionIndex = this.#currentQuestionIndex(); + const question = this.questions[questionIndex]; + if (!question) return lines; + if (lines.length > 0) lines.push(""); + lines.push(...renderQuestionTitle(question, questionIndex, width)); + return lines; + } + + #footerHintText(indicator: string): string { + const scroll = indicator ? ` ${indicator} scroll ·` : ""; + if (this.#isSubmitTab()) { + return `Enter submit · ↑/↓ move ·${scroll} Esc cancel`; + } + const question = this.questions[this.#currentQuestionIndex()]; + const action = question?.multi ? "Space/Enter toggle · n note · Next → continue" : "Enter select · n note"; + const tabs = this.#hasSubmitTab() ? " · Tab/←/→ tabs" : ""; + return `${action} · ↑/↓ move${tabs} ·${scroll} Esc cancel`; + } + + #questionRows(question: ExtensionAskDialogQuestion): QuestionRow[] { + const rows: QuestionRow[] = question.options.map((option, index) => ({ + kind: "option", + key: `option:${index}`, + label: this.#optionLabel(question, option.label, index), + optionIndex: index, + })); + rows.push({ kind: "other", key: "other", label: OTHER_OPTION, optionIndex: undefined }); + if (question.multi) rows.push({ kind: "next", key: "next", label: NEXT_OPTION, optionIndex: undefined }); + rows.push({ kind: "chat", key: "chat", label: CHAT_ABOUT_THIS_OPTION, optionIndex: undefined }); + return rows; + } + + #optionLabel(question: ExtensionAskDialogQuestion, label: string, index: number): string { + return question.recommended === index ? `${label} (Recommended)` : label; + } + + #activeQuestionState(): { question: ExtensionAskDialogQuestion; state: QuestionState } | undefined { + const question = this.questions[this.#currentQuestionIndex()]; + const state = this.#states[this.#currentQuestionIndex()]; + if (!question || !state) return undefined; + return { question, state }; + } + + #handleQuestionInput(keyData: string): void { + const active = this.#activeQuestionState(); + if (!active) return; + const { question, state } = active; + const rows = this.#questionRows(question); + if (matchesSelectUp(keyData)) { + state.cursorIndex = clamp(state.cursorIndex - 1, 0, Math.max(0, rows.length - 1)); + this.#requestRender(); + return; + } + if (matchesSelectDown(keyData)) { + state.cursorIndex = clamp(state.cursorIndex + 1, 0, Math.max(0, rows.length - 1)); + this.#requestRender(); + return; + } + const rowItem = rows[state.cursorIndex]; + if (!rowItem) return; + if (keyData === "n" || keyData === "N") { + if (rowItem.kind === "option" || rowItem.kind === "other") { + void this.#promptForNote(question, state, rowItem); + } + return; + } + const isEnter = matchesKey(keyData, "enter") || matchesKey(keyData, "return") || keyData === "\n"; + const isSpace = matchesKey(keyData, "space") || keyData === " "; + if (!isEnter && !(question.multi && isSpace)) return; + if (rowItem.kind === "chat") { + this.#finishChat(); + return; + } + if (rowItem.kind === "next") { + this.#advanceAfterQuestion(); + return; + } + if (rowItem.kind === "other") { + void this.#promptForCustomInput(question, state, rowItem); + return; + } + if (rowItem.kind === "option") { + const option = question.options[rowItem.optionIndex ?? -1]; + if (!option) return; + if (question.multi) { + if (state.selectedOptions.has(option.label)) { + state.selectedOptions.delete(option.label); + clearNoteIfRow(state, rowItem.key); + } else { + state.selectedOptions.add(option.label); + } + this.#requestRender(); + return; + } + state.selectedOptions = new Set([option.label]); + state.customInput = undefined; + clearNoteUnlessRow(state, rowItem.key); + this.#advanceAfterQuestion(); + } + } + + #handleSubmitTabInput(keyData: string): void { + const rows = this.#submitRows(); + if (matchesSelectUp(keyData)) { + this.#submitCursorIndex = clamp(this.#submitCursorIndex - 1, 0, Math.max(0, rows.length - 1)); + this.#requestRender(); + return; + } + if (matchesSelectDown(keyData)) { + this.#submitCursorIndex = clamp(this.#submitCursorIndex + 1, 0, Math.max(0, rows.length - 1)); + this.#requestRender(); + return; + } + const isEnter = matchesKey(keyData, "enter") || matchesKey(keyData, "return") || keyData === "\n"; + if (!isEnter) return; + const rowItem = rows[this.#submitCursorIndex]; + if (rowItem?.kind === "chat") { + this.#finishChat(); + return; + } + this.#finishSubmit(); + } + + #switchTab(direction: 1 | -1): void { + const tabCount = this.questions.length + 1; + this.#activeTabIndex = (this.#activeTabIndex + direction + tabCount) % tabCount; + this.#submitCursorIndex = 0; + } + + #advanceAfterQuestion(): void { + const current = this.#currentQuestionIndex(); + if (this.questions.length === 1) { + this.#finishSubmit(); + return; + } + this.#activeTabIndex = current + 1 < this.questions.length ? current + 1 : this.#submitTabIndex(); + this.#submitCursorIndex = 0; + this.#requestRender(); + } + + async #promptForCustomInput( + question: ExtensionAskDialogQuestion, + state: QuestionState, + rowItem: QuestionRow, + ): Promise { + this.#promptActive = true; + try { + const input = await this.callbacks.onPrompt(`Custom answer: ${question.question}`, state.customInput); + if (input === undefined || this.#closed) return; + state.customInput = input; + if (!question.multi) { + state.selectedOptions.clear(); + clearNoteUnlessRow(state, rowItem.key); + } + this.#advanceAfterQuestion(); + } finally { + this.#promptActive = false; + this.#runDeferredTimeout(); + this.#requestRender(); + } + } + + async #promptForNote( + question: ExtensionAskDialogQuestion, + state: QuestionState, + rowItem: QuestionRow, + ): Promise { + this.#promptActive = true; + try { + const input = await this.callbacks.onPrompt(`Note for ${rowItem.label}: ${question.question}`, state.note); + if (input === undefined || this.#closed) return; + state.note = input; + state.noteRowKey = rowItem.key; + } finally { + this.#promptActive = false; + this.#runDeferredTimeout(); + this.#requestRender(); + } + } + + #renderQuestionBody(width: number, rows: number): RenderedList { + const active = this.#activeQuestionState(); + if (!active) return { lines: Array.from({ length: rows }, () => ""), scrollOffset: 0, indicator: "" }; + const { question, state } = active; + const rowItems = this.#questionRows(question); + state.cursorIndex = clamp(state.cursorIndex, 0, Math.max(0, rowItems.length - 1)); + const selectedRow = rowItems[state.cursorIndex]; + const preview = + selectedRow?.kind === "option" ? question.options[selectedRow.optionIndex ?? -1]?.preview : undefined; + const sideBySide = width >= SIDE_BY_SIDE_LIST_MIN_WIDTH + PREVIEW_MIN_WIDTH + SIDE_BY_SIDE_GAP_WIDTH; + if (sideBySide) { + const previewWidth = Math.max(PREVIEW_MIN_WIDTH, Math.floor(width * 0.45)); + const listWidth = Math.max(1, width - previewWidth - SIDE_BY_SIDE_GAP_WIDTH); + const list = this.#renderQuestionList(question, state, rowItems, listWidth, rows); + const previewLines = this.#renderPreviewPane(preview, previewWidth, rows); + const lines: string[] = []; + for (let index = 0; index < rows; index++) { + const left = truncateToWidth(list.lines[index] ?? "", listWidth, Ellipsis.Unicode); + const right = truncateToWidth(previewLines[index] ?? "", previewWidth, Ellipsis.Unicode); + const gap = padding(Math.max(1, listWidth - visibleWidth(left)) + 1); + lines.push(`${left}${gap}${theme.fg("border", "│")} ${right}`); + } + return { lines, scrollOffset: list.scrollOffset, indicator: list.indicator }; + } + const previewRows = Math.max(3, Math.min(8, Math.floor(rows * 0.4))); + const listRows = Math.max(3, rows - previewRows - 1); + const list = this.#renderQuestionList(question, state, rowItems, width, listRows); + const previewLines = this.#renderPreviewPane(preview, width, previewRows); + const lines = [...list.lines, theme.fg("border", "─".repeat(Math.max(1, width))), ...previewLines]; + while (lines.length < rows) lines.push(""); + return { lines: lines.slice(0, rows), scrollOffset: list.scrollOffset, indicator: list.indicator }; + } + + #renderQuestionList( + question: ExtensionAskDialogQuestion, + state: QuestionState, + rowItems: QuestionRow[], + width: number, + rows: number, + ): RenderedList { + const mdTheme = getMarkdownTheme(); + const allLines: string[] = []; + const lineStartByRow: number[] = []; + for (let index = 0; index < rowItems.length; index++) { + lineStartByRow.push(allLines.length); + const rowItem = rowItems[index]; + if (!rowItem) continue; + allLines.push(...renderRowLabel(rowItem, question, state, index === state.cursorIndex, mdTheme, width)); + } + const cursorStart = lineStartByRow[state.cursorIndex] ?? 0; + state.scrollOffset = this.#scrollOffsetForCursor(state.scrollOffset, cursorStart, rows, allLines.length); + const scrollView = new ScrollView(allLines, { + height: rows, + scrollbar: "auto", + totalRows: allLines.length, + theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) }, + }); + scrollView.setScrollOffset(state.scrollOffset); + const rendered = scrollView.render(width); + const lines = [...rendered]; + while (lines.length < rows) lines.push(""); + return { + lines: lines.slice(0, rows), + scrollOffset: state.scrollOffset, + indicator: this.#clipIndicator(state.scrollOffset, rows, allLines.length), + }; + } + + #renderPreviewPane(preview: string | undefined, width: number, rows: number): string[] { + const bodyWidth = Math.max(1, width - 2); + const out = [theme.fg("dim", PREVIEW_HEADER)]; + const contentRows = Math.max(0, rows - 1); + const content = renderPreviewContent(preview, bodyWidth); + const hidden = Math.max(0, content.length - contentRows); + const visibleCount = hidden > 0 ? Math.max(0, contentRows - 1) : Math.min(contentRows, content.length); + for (let index = 0; index < visibleCount; index++) out.push(content[index] ?? ""); + if (hidden > 0) out.push(theme.fg("dim", `… ${hidden + 1} more lines`)); + while (out.length < rows) out.push(""); + return out.slice(0, rows); + } + + #renderSubmitBody(width: number, rows: number): RenderedList { + const allLines: string[] = []; + const unanswered = this.#unansweredCount(); + if (unanswered > 0) { + allLines.push( + theme.fg( + "warning", + `${unanswered} unanswered question${unanswered === 1 ? "" : "s"}; Enter still submits.`, + ), + ); + allLines.push(""); + } + for (let index = 0; index < this.questions.length; index++) { + const question = this.questions[index]; + const state = this.#states[index]; + if (!question || !state) continue; + const label = questionTabLabel(question, index); + const answer = renderAnswerSummary(question, state); + allLines.push(`${theme.fg("dim", `${index + 1}. ${label}:`)} ${answer}`); + const submittedNote = noteForSubmittedAnswer(question, state); + if (submittedNote?.trim()) { + const note = normalizedInlineInput(submittedNote); + allLines.push( + theme.fg("muted", ` Note: ${truncateToWidth(note, Math.max(1, width - 9), Ellipsis.Unicode)}`), + ); + } + } + allLines.push(""); + const rowStart = allLines.length; + const submitRows = this.#submitRows(); + for (let index = 0; index < submitRows.length; index++) { + const rowItem = submitRows[index]; + if (!rowItem) continue; + const cursor = index === this.#submitCursorIndex ? theme.fg("accent", `${theme.nav.cursor} `) : " "; + const color = index === this.#submitCursorIndex ? "accent" : "text"; + allLines.push(`${cursor}${theme.fg(color, rowItem.label)}`); + } + const cursorStart = rowStart + this.#submitCursorIndex; + this.#submitScrollOffset = this.#scrollOffsetForCursor( + this.#submitScrollOffset, + cursorStart, + rows, + allLines.length, + ); + const scrollView = new ScrollView(allLines, { + height: rows, + scrollbar: "auto", + totalRows: allLines.length, + theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) }, + }); + scrollView.setScrollOffset(this.#submitScrollOffset); + const rendered = scrollView.render(width); + const lines = [...rendered]; + while (lines.length < rows) lines.push(""); + return { + lines: lines.slice(0, rows), + scrollOffset: this.#submitScrollOffset, + indicator: this.#clipIndicator(this.#submitScrollOffset, rows, allLines.length), + }; + } + + #submitRows(): SubmitRow[] { + return [ + { kind: "submit", key: "submit", label: SUBMIT_OPTION }, + { kind: "chat", key: "chat", label: CHAT_ABOUT_THIS_OPTION }, + ]; + } + + #scrollOffsetForCursor(currentOffset: number, cursorLine: number, rows: number, totalRows: number): number { + if (totalRows <= rows) return 0; + let nextOffset = clamp(currentOffset, 0, Math.max(0, totalRows - rows)); + if (cursorLine < nextOffset) nextOffset = cursorLine; + if (cursorLine >= nextOffset + rows) nextOffset = cursorLine - rows + 1; + return clamp(nextOffset, 0, Math.max(0, totalRows - rows)); + } + + #clipIndicator(offset: number, rows: number, totalRows: number): string { + const above = offset > 0; + const below = offset + rows < totalRows; + if (above && below) return "↕"; + if (above) return "↑"; + if (below) return "↓"; + return ""; + } + + #unansweredCount(): number { + let count = 0; + for (let index = 0; index < this.questions.length; index++) { + const question = this.questions[index]; + const state = this.#states[index]; + if (!question || !state) continue; + if (state.selectedOptions.size === 0 && state.customInput === undefined) count += 1; + } + return count; + } + + #handleTimeout(): void { + if (this.#closed) return; + if (this.#promptActive) { + this.#timeoutExpired = true; + return; + } + this.options.onTimeout?.(); + for (let index = 0; index < this.questions.length; index++) { + const question = this.questions[index]; + const state = this.#states[index]; + if (!question || !state) continue; + if (state.selectedOptions.size === 0 && state.customInput === undefined) { + const noteMatch = /^option:(\d+)$/.exec(state.noteRowKey ?? ""); + const notedIndex = noteMatch ? Number.parseInt(noteMatch[1], 10) : Number.NaN; + const fallbackIndex = + Number.isInteger(notedIndex) && question.options[notedIndex] + ? notedIndex + : clamp(question.recommended ?? 0, 0, Math.max(0, question.options.length - 1)); + const fallback = question.options[fallbackIndex]; + if (fallback) state.selectedOptions.add(fallback.label); + state.timedOut = true; + } + } + this.#finishSubmit(); + } + + #runDeferredTimeout(): void { + if (!this.#timeoutExpired) return; + this.#timeoutExpired = false; + this.#handleTimeout(); + } + + #finishSubmit(): void { + if (this.#closed) return; + this.#closed = true; + this.#countdown?.dispose(); + this.callbacks.onSubmit({ kind: "submit", results: this.#buildResults() }); + } + + #finishCancel(): void { + if (this.#closed) return; + this.#closed = true; + this.#countdown?.dispose(); + this.callbacks.onCancel(); + } + + #finishChat(): void { + if (this.#closed) return; + this.#closed = true; + this.#countdown?.dispose(); + this.callbacks.onChat(); + } + + #buildResults(): ExtensionAskDialogResultItem[] { + const results: ExtensionAskDialogResultItem[] = []; + for (let index = 0; index < this.questions.length; index++) { + const question = this.questions[index]; + const state = this.#states[index]; + if (!question || !state) continue; + const selectedOptions = question.options + .map(option => option.label) + .filter(label => state.selectedOptions.has(label)); + results.push({ + id: question.id, + question: question.question, + options: question.options.map(option => option.label), + multi: question.multi ?? false, + selectedOptions, + customInput: state.customInput, + note: noteForSubmittedAnswer(question, state), + timedOut: state.timedOut || undefined, + }); + } + return results; + } +} diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index 1c9172544..3fd1b5723 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -5,6 +5,9 @@ import { KeybindingsManager } from "../../config/keybindings"; import type { CompactOptions, ExtensionActions, + ExtensionAskDialogQuestion, + ExtensionAskDialogResult, + ExtensionAskDialogResultItem, ExtensionCommandContextActions, ExtensionContextActions, ExtensionError, @@ -19,6 +22,7 @@ import type { } from "../../extensibility/extensions"; import { getSessionSlashCommands } from "../../extensibility/extensions/get-commands-handler"; import { createExtensionModelQuery } from "../../extensibility/extensions/model-api"; +import { AskDialogComponent } from "../../modes/components/ask-dialog"; import { HookEditorComponent } from "../../modes/components/hook-editor"; import { HookInputComponent } from "../../modes/components/hook-input"; import { HookSelectorComponent, type HookSelectorSlider } from "../../modes/components/hook-selector"; @@ -28,12 +32,20 @@ import { USER_INTERRUPT_LABEL } from "../../session/messages"; import { setSessionTerminalTitle, setTerminalTitle } from "../../utils/title-generator"; const MAX_WIDGET_LINES = 10; +const ASK_OTHER_OPTION = "Other (type your own)"; +const ASK_CHAT_OPTION = "Chat about this"; +const ASK_NEXT_OPTION = "Next →"; interface CollabDialogWinner { source: "local" | "remote"; value: string | undefined; } +interface CollabAskDialogWinner { + source: "local" | "remote"; + value: ExtensionAskDialogResult | undefined; +} + function toWireSelectOptions(options: ExtensionUISelectItem[]): CollabUiSelectItem[] { return options.map(option => typeof option === "string" @@ -64,6 +76,7 @@ export class ExtensionUiController { select: (title, options, dialogOptions) => this.showCollabAwareSelector(title, options, dialogOptions), confirm: (title, message, _dialogOptions) => this.showHookConfirm(title, message), input: (title, placeholder, dialogOptions) => this.showHookInput(title, placeholder, dialogOptions), + askDialog: (questions, dialogOptions) => this.showAskDialog(questions, dialogOptions), notify: (message, type) => this.showHookNotify(message, type), onTerminalInput: handler => this.addExtensionTerminalInputListener(handler), setStatus: (key, text) => this.setHookStatus(key, text), @@ -579,6 +592,107 @@ export class ExtensionUiController { ); } + async showAskDialog( + questions: ExtensionAskDialogQuestion[], + dialogOptions?: ExtensionUIDialogOptions, + ): Promise { + const host = this.ctx.collabHost; + if (!host) return this.#showLocalAskDialog(questions, dialogOptions); + const localAbort = new AbortController(); + const remoteAbort = new AbortController(); + const parentSignal = dialogOptions?.signal; + const localSignal = parentSignal ? AbortSignal.any([parentSignal, localAbort.signal]) : localAbort.signal; + const remoteSignal = parentSignal ? AbortSignal.any([parentSignal, remoteAbort.signal]) : remoteAbort.signal; + const localWinner = this.#showLocalAskDialog(questions, { ...dialogOptions, signal: localSignal }).then( + (value): CollabAskDialogWinner => ({ source: "local", value }), + ); + const remoteWinner: Promise = this.#runGuestAskDialog(questions, remoteSignal).then( + result => (result === "unavailable" ? localWinner : { source: "remote", value: result }), + ); + const winner = await Promise.race([localWinner, remoteWinner]); + if (winner.source === "remote") localAbort.abort(); + else remoteAbort.abort(); + return winner.value; + } + + #showLocalAskDialog( + questions: ExtensionAskDialogQuestion[], + dialogOptions?: ExtensionUIDialogOptions, + ): Promise { + return this.#presentDialog(dialogOptions?.signal, settle => { + let askDialog: AskDialogComponent | undefined; + let promptEditor: HookEditorComponent | undefined; + let promptResolve: ((value: string | undefined) => void) | undefined; + let closed = false; + + const restoreAskDialog = (): void => { + if (closed || !askDialog) return; + this.ctx.editorContainer.clear(); + this.ctx.editorContainer.addChild(askDialog); + this.ctx.ui.setFocus(askDialog); + this.ctx.ui.requestRender(); + }; + + const finishPrompt = (value: string | undefined): void => { + const resolvePrompt = promptResolve; + promptResolve = undefined; + promptEditor = undefined; + restoreAskDialog(); + resolvePrompt?.(value); + }; + + const promptForText = (title: string, prefill?: string): Promise => { + if (closed) return Promise.resolve(undefined); + const { promise, resolve } = Promise.withResolvers(); + promptResolve = resolve; + promptEditor = new HookEditorComponent( + this.ctx.ui, + title, + prefill, + value => finishPrompt(value), + () => finishPrompt(undefined), + { promptStyle: true }, + ); + this.ctx.editorContainer.clear(); + this.ctx.editorContainer.addChild(promptEditor); + this.ctx.ui.setFocus(promptEditor); + this.ctx.ui.requestRender(); + return promise; + }; + + askDialog = new AskDialogComponent( + questions, + { + onSubmit: result => settle(result), + onCancel: () => settle(undefined), + onChat: () => settle(undefined), + onPrompt: promptForText, + }, + { + timeout: dialogOptions?.timeout, + onTimeout: dialogOptions?.onTimeout, + tui: this.ctx.ui, + }, + ); + this.ctx.editorContainer.clear(); + this.ctx.editorContainer.addChild(askDialog); + this.ctx.ui.setFocus(askDialog); + this.ctx.ui.requestRender(); + + return () => { + closed = true; + askDialog?.dispose(); + promptResolve?.(undefined); + promptResolve = undefined; + promptEditor = undefined; + this.ctx.editorContainer.clear(); + this.ctx.editorContainer.addChild(this.ctx.editor); + this.ctx.ui.setFocus(this.ctx.editor); + this.ctx.ui.requestRender(); + }; + }); + } + /** * Race the local hook dialog against a mirrored guest ask. First *answer* * wins and cancels the other side. A remote `unavailable` settlement @@ -612,6 +726,114 @@ export class ExtensionUiController { return winner.value; } + async #runGuestAskDialog( + questions: ExtensionAskDialogQuestion[], + signal: AbortSignal, + ): Promise { + const results: ExtensionAskDialogResultItem[] = []; + for (const question of questions) { + const result = await this.#runGuestAskQuestion(question, signal); + if (result === "unavailable" || result === undefined) return result; + results.push(result); + } + return { kind: "submit", results }; + } + + async #runGuestAskQuestion( + question: ExtensionAskDialogQuestion, + signal: AbortSignal, + ): Promise { + const selected = new Set(); + let customInput: string | undefined; + const baseOptions: CollabUiSelectItem[] = question.options.map(option => + option.description?.trim() ? { label: option.label, description: option.description.trim() } : option.label, + ); + if (question.multi) { + while (true) { + const checkedIndices = question.options + .map((option, index) => (selected.has(option.label) ? index : -1)) + .filter(index => index >= 0); + const choice = await this.#requestGuestUiString( + { + kind: "select", + title: question.question, + options: [...baseOptions, ASK_OTHER_OPTION, ASK_NEXT_OPTION, ASK_CHAT_OPTION], + selectionMarker: "checkbox", + checkedIndices, + markableCount: question.options.length, + helpText: "up/down navigate enter toggle Next → continue esc cancel", + }, + signal, + ); + if (choice === "unavailable" || choice === undefined) return choice; + if (choice === ASK_CHAT_OPTION) return undefined; + if (choice === ASK_NEXT_OPTION) break; + if (choice === ASK_OTHER_OPTION) { + const input = await this.#requestGuestUiString( + { kind: "editor", title: `Custom answer: ${question.question}` }, + signal, + ); + if (input === "unavailable" || input === undefined) return input; + customInput = input; + break; + } + if (selected.has(choice)) selected.delete(choice); + else selected.add(choice); + } + } else { + const recommended = + typeof question.recommended === "number" && Number.isInteger(question.recommended) + ? question.recommended + : 0; + const initialIndex = Math.max(0, Math.min(recommended, Math.max(0, question.options.length - 1))); + const choice = await this.#requestGuestUiString( + { + kind: "select", + title: question.question, + options: [...baseOptions, ASK_OTHER_OPTION, ASK_CHAT_OPTION], + initialIndex, + selectionMarker: "radio", + markableCount: question.options.length, + helpText: "up/down navigate enter select esc cancel", + }, + signal, + ); + if (choice === "unavailable" || choice === undefined) return choice; + if (choice === ASK_CHAT_OPTION) return undefined; + if (choice === ASK_OTHER_OPTION) { + const input = await this.#requestGuestUiString( + { kind: "editor", title: `Custom answer: ${question.question}` }, + signal, + ); + if (input === "unavailable" || input === undefined) return input; + customInput = input; + } else { + selected.add(choice); + } + } + return { + id: question.id, + question: question.question, + options: question.options.map(option => option.label), + multi: question.multi ?? false, + selectedOptions: question.options.map(option => option.label).filter(label => selected.has(label)), + customInput, + }; + } + + async #requestGuestUiString( + request: CollabUiRequestDraft, + signal: AbortSignal, + ): Promise { + const host = this.ctx.collabHost; + if (!host) return "unavailable"; + const remote = host.requestGuestUi(request, signal); + if (!remote) return "unavailable"; + const result = await remote; + if (result.kind === "unavailable") return "unavailable"; + return typeof result.value === "string" ? result.value : undefined; + } + /** * Show a selector for hooks. */ @@ -914,11 +1136,11 @@ export class ExtensionUiController { * the current dialog and hands the surface to the next queued request. A request * whose signal aborts before its turn resolves `undefined` and is never shown. */ - #presentDialog( + #presentDialog( signal: AbortSignal | undefined, - present: (settle: (value: string | undefined) => void) => () => void, - ): Promise { - const { promise, resolve, reject } = Promise.withResolvers(); + present: (settle: (value: T | undefined) => void) => () => void, + ): Promise { + const { promise, resolve, reject } = Promise.withResolvers(); let settled = false; let started = false; let hide: (() => void) | undefined; @@ -927,7 +1149,7 @@ export class ExtensionUiController { settle(undefined); } - const settle = (value: string | undefined): void => { + const settle = (value: T | undefined): void => { if (settled) return; settled = true; signal?.removeEventListener("abort", onAbort); diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index 42de7f930..d6663962c 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -23,6 +23,7 @@ import { Markdown, type MarkdownTheme, renderInlineMarkdown, + replaceTabs, TERMINAL, Text, truncateToWidth, @@ -44,17 +45,34 @@ import { ToolAbortError } from "./tool-errors"; // Types // ============================================================================= +const OTHER_OPTION = "Other (type your own)"; +const CHAT_ABOUT_THIS_OPTION = "Chat about this"; +const NEXT_OPTION = "Next →"; +const RESERVED_OPTION_LABELS: Record = { + [OTHER_OPTION]: true, + [CHAT_ABOUT_THIS_OPTION]: true, + [NEXT_OPTION]: true, +}; + const OptionItem = arkType({ label: arkType("string").describe("display label"), "description?": arkType("string").describe("optional explanatory text displayed below the label"), + "preview?": arkType("string").describe("optional rich preview content for interactive ask dialogs"), }); const QuestionItem = arkType({ id: arkType("string").describe("question id"), question: arkType("string").describe("question text"), + "header?": arkType("string").describe("optional short display chip for rich ask dialogs"), options: OptionItem.array().describe("available options"), "multi?": arkType("boolean").describe("allow multiple selections"), "recommended?": arkType("number").describe("recommended option index"), +}).narrow((question, ctx) => { + const reserved = question.options.find(option => RESERVED_OPTION_LABELS[option.label] === true); + return ( + reserved === undefined || + ctx.mustBe(`defined with option labels that do not collide with reserved runtime labels: ${reserved.label}`) + ); }); const askSchema = arkType({ @@ -71,6 +89,8 @@ export interface QuestionResult { multi: boolean; selectedOptions: string[]; customInput?: string; + /** Optional note attached to the selected answer in the rich ask dialog. */ + note?: string; /** True when the answer was auto-selected because the dialog timed out. */ timedOut?: boolean; } @@ -81,6 +101,8 @@ export interface AskToolDetails { multi?: boolean; selectedOptions?: string[]; customInput?: string; + /** Optional note attached to the selected answer in the rich ask dialog. */ + note?: string; /** True when the answer was auto-selected because the dialog timed out. */ timedOut?: boolean; /** Multi-part question mode */ @@ -108,7 +130,6 @@ function toSelectOption(option: AskOption, label = option.label): ExtensionUISel // Constants // ============================================================================= -const OTHER_OPTION = "Other (type your own)"; const RECOMMENDED_SUFFIX = " (Recommended)"; // Window after the timeout deadline within which an `undefined` selection is // attributed to a UI-enforced timeout (for surfaces that close the dialog at @@ -361,6 +382,7 @@ function formatCustomInputTitle( interface SelectionResult { selectedOptions: string[]; customInput?: string; + note?: string; timedOut: boolean; navigation?: "back" | "forward"; cancelled?: boolean; @@ -375,7 +397,7 @@ interface AskSingleQuestionOptions { recommended?: number; timeout?: number; signal?: AbortSignal; - initialSelection?: Pick; + initialSelection?: Pick; navigation?: NavigationControls; } @@ -416,6 +438,7 @@ async function askSingleQuestion( const doneLabel = getDoneOptionLabel(); let selectedOptions = [...(initialSelection?.selectedOptions ?? [])]; let customInput = initialSelection?.customInput; + const note = initialSelection?.note; let timedOut = false; const selectOption = async ( @@ -513,14 +536,14 @@ async function askSingleQuestion( }); if (arrowNavigation) { - return { selectedOptions: Array.from(selected), customInput, timedOut, navigation: arrowNavigation }; + return { selectedOptions: Array.from(selected), customInput, note, timedOut, navigation: arrowNavigation }; } if (choice === undefined) { if (selectTimedOut) { timedOut = true; break; } - return { selectedOptions: Array.from(selected), customInput, timedOut, cancelled: true }; + return { selectedOptions: Array.from(selected), customInput, note, timedOut, cancelled: true }; } if (choice === doneLabel) break; @@ -587,11 +610,11 @@ async function askSingleQuestion( timedOut = selectTimedOut; if (arrowNavigation) { - return { selectedOptions, customInput, timedOut, navigation: arrowNavigation }; + return { selectedOptions, customInput, note, timedOut, navigation: arrowNavigation }; } if (choice === undefined) { if (!timedOut) { - return { selectedOptions, customInput, timedOut, cancelled: true }; + return { selectedOptions, customInput, note, timedOut, cancelled: true }; } break; } @@ -615,7 +638,7 @@ async function askSingleQuestion( break; } if (navigation?.allowForward) { - return { selectedOptions, customInput, timedOut, navigation: "forward" }; + return { selectedOptions, customInput, note, timedOut, navigation: "forward" }; } } @@ -623,20 +646,58 @@ async function askSingleQuestion( selectedOptions = getAutoSelectionOnTimeout(questionOptions, recommended); } - return { selectedOptions, customInput, timedOut }; + return { selectedOptions, customInput, note, timedOut }; } function formatQuestionResult(result: QuestionResult): string { + const noteSuffix = result.note ? ` (note: ${result.note})` : ""; if (result.customInput !== undefined) { - return `${result.id}: "${result.customInput}"`; + return `${result.id}: "${result.customInput}"${noteSuffix}`; } if (result.selectedOptions.length > 0) { - const suffix = result.timedOut ? " (auto-selected after timeout)" : ""; + const suffix = `${result.timedOut ? " (auto-selected after timeout)" : ""}${noteSuffix}`; return result.multi ? `${result.id}: [${result.selectedOptions.join(", ")}]${suffix}` : `${result.id}: ${result.selectedOptions[0]}${suffix}`; } - return `${result.id}: (cancelled)`; + return `${result.id}: (cancelled)${noteSuffix}`; +} + +function formatSingleQuestionResponse(result: { + selectedOptions: string[]; + customInput?: string; + note?: string; + timedOut?: boolean; + multi: boolean; +}): string { + const responseParts: string[] = []; + if (result.selectedOptions.length > 0) { + const selectedText = result.multi + ? `User selected: ${result.selectedOptions.join(", ")}` + : `User selected: ${result.selectedOptions[0]}`; + responseParts.push(result.timedOut ? `${selectedText} (auto-selected after timeout)` : selectedText); + } + if (result.customInput !== undefined) { + responseParts.push( + result.customInput.includes("\n") + ? `User provided custom input:\n${result.customInput + .split("\n") + .map(line => ` ${line}`) + .join("\n")}` + : `User provided custom input: ${result.customInput}`, + ); + } + if (result.note) { + responseParts.push( + result.note.includes("\n") + ? `User added note:\n${result.note + .split("\n") + .map(line => ` ${line}`) + .join("\n")}` + : `User added note: ${result.note}`, + ); + } + return responseParts.length > 0 ? responseParts.join("\n") : "User cancelled the selection"; } // ============================================================================= @@ -772,6 +833,83 @@ export class AskTool implements AgentTool { vocalizer.speak(params.questions.map(q => q.question).join("\n")); } + const richAskDialog = extensionUi.askDialog; + if (richAskDialog) { + try { + const showRichDialog = () => + richAskDialog( + params.questions.map(q => ({ + id: q.id, + question: q.question, + ...(q.header?.trim() ? { header: q.header } : {}), + options: q.options.map(option => ({ + label: option.label, + ...(option.description?.trim() ? { description: option.description.trim() } : {}), + ...(option.preview?.trim() ? { preview: option.preview } : {}), + })), + ...(q.multi !== undefined ? { multi: q.multi } : {}), + ...(q.recommended !== undefined ? { recommended: q.recommended } : {}), + })), + { timeout: timeout ?? undefined, signal }, + ); + const richResult = signal ? await untilAborted(signal, showRichDialog) : await showRichDialog(); + if (!richResult) { + context.abort(); + throw new ToolAbortError("Ask tool was cancelled by the user"); + } + if (richResult.results.length !== params.questions.length) { + throw new Error("Ask dialog returned a result count that does not match the requested questions"); + } + const results: QuestionResult[] = []; + for (let index = 0; index < params.questions.length; index++) { + const question = params.questions[index]; + const result = richResult.results[index]; + if (!question || !result || result.id !== question.id) { + throw new Error("Ask dialog returned results that do not match the requested question order"); + } + results.push({ + id: question.id, + question: question.question, + options: question.options.map(option => option.label), + multi: question.multi ?? false, + selectedOptions: result.selectedOptions, + customInput: result.customInput, + note: result.note, + timedOut: result.timedOut, + }); + } + if (params.questions.length === 1) { + const result = results[0]; + if ( + !result || + (!result.timedOut && result.selectedOptions.length === 0 && result.customInput === undefined) + ) { + context.abort(); + throw new ToolAbortError("Ask tool was cancelled by the user"); + } + const details: AskToolDetails = { + question: result.question, + options: result.options, + multi: result.multi, + selectedOptions: result.selectedOptions, + customInput: result.customInput, + note: result.note, + timedOut: result.timedOut, + }; + const responseText = formatSingleQuestionResponse(result); + return { content: [{ type: "text" as const, text: responseText }], details }; + } + const details: AskToolDetails = { results }; + const responseText = `User answers:\n${results.map(formatQuestionResult).join("\n")}`; + return { content: [{ type: "text" as const, text: responseText }], details }; + } catch (error) { + if (error instanceof Error && error.name === "AbortError") { + throw new ToolAbortError("Ask input was cancelled"); + } + throw error; + } + } + const askQuestion = async ( q: AskParams["questions"][number], options?: { previous?: QuestionResult; navigation?: NavigationControls }, @@ -782,7 +920,7 @@ export class AskTool implements AgentTool { })); const optionLabels = questionOptions.map(getAskOptionLabel); try { - const { selectedOptions, customInput, navigation, cancelled, timedOut } = await askSingleQuestion( + const { selectedOptions, customInput, note, navigation, cancelled, timedOut } = await askSingleQuestion( ui, q.question, questionOptions, @@ -795,7 +933,7 @@ export class AskTool implements AgentTool { navigation: options?.navigation, }, ); - return { optionLabels, selectedOptions, customInput, navigation, cancelled, timedOut }; + return { optionLabels, selectedOptions, customInput, note, navigation, cancelled, timedOut }; } catch (error) { if (error instanceof Error && error.name === "AbortError") { throw new ToolAbortError("Ask input was cancelled"); @@ -806,7 +944,7 @@ export class AskTool implements AgentTool { if (params.questions.length === 1) { const [q] = params.questions; - const { optionLabels, selectedOptions, customInput, cancelled, timedOut } = await askQuestion(q); + const { optionLabels, selectedOptions, customInput, note, cancelled, timedOut } = await askQuestion(q); if (!timedOut && (cancelled || (selectedOptions.length === 0 && customInput === undefined))) { context.abort(); @@ -818,27 +956,17 @@ export class AskTool implements AgentTool { multi: q.multi ?? false, selectedOptions, customInput, + note, timedOut: timedOut || undefined, }; - const responseParts: string[] = []; - if (selectedOptions.length > 0) { - const selectedText = q.multi - ? `User selected: ${selectedOptions.join(", ")}` - : `User selected: ${selectedOptions[0]}`; - responseParts.push(timedOut ? `${selectedText} (auto-selected after timeout)` : selectedText); - } - if (customInput !== undefined) { - responseParts.push( - customInput.includes("\n") - ? `User provided custom input:\n${customInput - .split("\n") - .map(line => ` ${line}`) - .join("\n")}` - : `User provided custom input: ${customInput}`, - ); - } - const responseText = responseParts.length > 0 ? responseParts.join("\n") : "User cancelled the selection"; + const responseText = formatSingleQuestionResponse({ + selectedOptions, + customInput, + note, + timedOut: timedOut || undefined, + multi: q.multi ?? false, + }); return { content: [{ type: "text" as const, text: responseText }], details }; } @@ -846,7 +974,8 @@ export class AskTool implements AgentTool { const resultsByIndex: Array = Array.from({ length: params.questions.length }); let questionIndex = 0; while (questionIndex < params.questions.length) { - const q = params.questions[questionIndex]!; + const q = params.questions[questionIndex]; + if (!q) throw new Error("Ask question index exceeded the requested question list"); const previous = resultsByIndex[questionIndex]; const navigation: NavigationControls = { allowBack: questionIndex > 0, @@ -857,6 +986,7 @@ export class AskTool implements AgentTool { optionLabels, selectedOptions, customInput, + note, navigation: navAction, cancelled, timedOut, @@ -874,6 +1004,7 @@ export class AskTool implements AgentTool { multi: q.multi ?? false, selectedOptions, customInput, + note, timedOut: timedOut || undefined, }; @@ -885,9 +1016,9 @@ export class AskTool implements AgentTool { questionIndex += 1; } - const results = resultsByIndex.map((result, index) => { + const results = params.questions.map((q, index) => { + const result = resultsByIndex[index]; if (result) return result; - const q = params.questions[index]!; return { id: q.id, question: q.question, @@ -986,6 +1117,21 @@ function renderCustomInputLines(uiTheme: Theme, customInput: string): string[] { return out; } +/** Render an answer note with tab replacement and line-width clamping. */ +function renderNoteLines(uiTheme: Theme, note: string, width: number): string[] { + const prefix = " Note: "; + const continuationPrefix = " "; + const firstLineWidth = Math.max(1, width - visibleWidth(prefix)); + const continuationWidth = Math.max(1, width - visibleWidth(continuationPrefix)); + return replaceTabs(note) + .split("\n") + .map((line, index) => { + const linePrefix = index === 0 ? `${uiTheme.fg("dim", " Note:")} ` : continuationPrefix; + const maxWidth = index === 0 ? firstLineWidth : continuationWidth; + return `${linePrefix}${uiTheme.fg("toolOutput", truncateToWidth(line, maxWidth))}`; + }); +} + /** * Marker glyph for a question option. Single-choice questions render circular radio * buttons (pick one); multi-select questions render rectangular checkboxes (pick many). @@ -1026,6 +1172,8 @@ function renderAnswerOptionLines( selectedOptions: string[] | undefined, multi: boolean | undefined, customInput: string | undefined, + note: string | undefined, + width: number, ): string[] { const selected = new Set(selectedOptions ?? []); // Prefer the full recorded option set; fall back to the selected labels when @@ -1033,7 +1181,7 @@ function renderAnswerOptionLines( const list = options && options.length > 0 ? options : (selectedOptions ?? []); // Nothing was chosen (and no custom answer) → a lone cancelled marker. - if (selected.size === 0 && customInput === undefined) { + if (selected.size === 0 && customInput === undefined && note === undefined) { return [` ${uiTheme.styledSymbol("status.warning", "warning")} ${uiTheme.fg("warning", "Cancelled")}`]; } @@ -1048,6 +1196,7 @@ function renderAnswerOptionLines( out.push(` ${markerStyled} ${labelStyled}`); } if (customInput !== undefined) out.push(...renderCustomInputLines(uiTheme, customInput)); + if (note !== undefined) out.push(...renderNoteLines(uiTheme, note, width)); return out; } @@ -1141,7 +1290,10 @@ export const askToolRenderer = { if (details.results && details.results.length > 0) { const results = details.results; const hasAnySelection = results.some( - r => r.customInput !== undefined || (r.selectedOptions && r.selectedOptions.length > 0), + r => + r.customInput !== undefined || + r.note !== undefined || + (r.selectedOptions && r.selectedOptions.length > 0), ); const header = renderStatusLine( { @@ -1156,7 +1308,16 @@ export const askToolRenderer = { // md() returns a shared cached array (module-level Markdown LRU) — copy before appending. const lines = [ ...md(r.question, width), - ...renderAnswerOptionLines(uiTheme, mdTheme, r.options, r.selectedOptions, r.multi, r.customInput), + ...renderAnswerOptionLines( + uiTheme, + mdTheme, + r.options, + r.selectedOptions, + r.multi, + r.customInput, + r.note, + width, + ), ]; return { label: uiTheme.fg("dim", `[${r.id}]`), lines }; }); @@ -1179,7 +1340,9 @@ export const askToolRenderer = { const question = details.question; const hasSelection = - details.customInput !== undefined || (details.selectedOptions && details.selectedOptions.length > 0); + details.customInput !== undefined || + details.note !== undefined || + (details.selectedOptions && details.selectedOptions.length > 0); const header = renderStatusLine( hasSelection ? { iconOverride: uiTheme.styledSymbol("tool.ask", "accent"), title: "Ask" } @@ -1190,12 +1353,13 @@ export const askToolRenderer = { const dSelected = details.selectedOptions; const dMulti = details.multi; const dCustom = details.customInput; + const dNote = details.note; const dTimedOut = details.timedOut; return framedBlock(uiTheme, width => { // md() returns a shared cached array (module-level Markdown LRU) — copy before appending. const bodyLines = [ ...md(question, width), - ...renderAnswerOptionLines(uiTheme, mdTheme, dOptions, dSelected, dMulti, dCustom), + ...renderAnswerOptionLines(uiTheme, mdTheme, dOptions, dSelected, dMulti, dCustom, dNote, width), ]; if (dTimedOut) { // Distinguish auto-selection from a real user choice in the transcript. diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 0cb1d4b3e..fb856e835 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -612,6 +612,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P if (name === "web_search") return session.settings.get("web_search.enabled"); // search_tool_bm25 is allowed when either legacy mcp.discoveryMode or new tools.discoveryMode is active. if (name === "search_tool_bm25") return discoveryActive; + if (name === "ask") return session.settings.get("ask.enabled"); if (name === "browser") return session.settings.get("browser.enabled"); if (name === "checkpoint" || name === "rewind") return session.settings.get("checkpoint.enabled"); if (name === "irc") return isIrcEnabled(session.settings, session.taskDepth ?? 0); diff --git a/packages/coding-agent/test/modes/components/ask-dialog.test.ts b/packages/coding-agent/test/modes/components/ask-dialog.test.ts new file mode 100644 index 000000000..f08b67268 --- /dev/null +++ b/packages/coding-agent/test/modes/components/ask-dialog.test.ts @@ -0,0 +1,738 @@ +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; +import { stripVTControlCharacters } from "node:util"; +import { KeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; +import type { ExtensionAskDialogQuestion } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; +import { AskDialogComponent } from "@oh-my-pi/pi-coding-agent/modes/components/ask-dialog"; +import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { setKeybindings } from "@oh-my-pi/pi-tui"; + +const DOWN = "\x1b[B"; +const ENTER = "\n"; +const CANCEL = "\x07"; +const SPACE = " "; +const TAB = "\t"; +const SHIFT_TAB = "\x1b[Z"; + +let darkTheme = await getThemeByName("dark"); + +function render(component: AskDialogComponent): string { + return stripVTControlCharacters(component.render(80).join("\n")); +} + +describe("AskDialogComponent", () => { + beforeAll(async () => { + darkTheme = await getThemeByName("dark"); + if (!darkTheme) throw new Error("Failed to load dark theme"); + }); + + beforeEach(() => { + setThemeInstance(darkTheme!); + setKeybindings(KeybindingsManager.inMemory({ "tui.select.cancel": "ctrl+g" })); + }); + + afterEach(() => { + setKeybindings(KeybindingsManager.inMemory()); + vi.useRealTimers(); + vi.restoreAllMocks(); + }); + + it("single-question, single-select: Enter on option submits immediately", () => { + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const onChat = vi.fn(); + const onPrompt = vi.fn(); + + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat, + onPrompt, + }); + + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0]).toEqual({ + kind: "submit", + results: [ + { + id: "q1", + question: "Choose one?", + options: ["Option A", "Option B"], + multi: false, + selectedOptions: ["Option A"], + customInput: undefined, + note: undefined, + timedOut: undefined, + }, + ], + }); + }); + + it("single-question, single-select: DOWN then Enter selects second option and submits", () => { + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const onChat = vi.fn(); + const onPrompt = vi.fn(); + + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat, + onPrompt, + }); + + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option B"]); + }); + + it("multi-question, single-select: Enter on option advances tab, does not submit", () => { + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const onChat = vi.fn(); + const onPrompt = vi.fn(); + + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Q1?", + options: [{ label: "A1" }, { label: "B1" }], + }, + { + id: "q2", + question: "Q2?", + options: [{ label: "A2" }, { label: "B2" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat, + onPrompt, + }); + + // Press Enter on A1 - should advance tab to Q2 (tab 1), not submit + component.handleInput(ENTER); + expect(onSubmit).not.toHaveBeenCalled(); + + // On Q2: Down to B2 and Enter - should advance tab to Submit (tab 2), not submit + component.handleInput(DOWN); + component.handleInput(ENTER); + expect(onSubmit).not.toHaveBeenCalled(); + + // On Submit tab: Enter on Submit row - should submit + component.handleInput(ENTER); + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results).toEqual([ + { + id: "q1", + question: "Q1?", + options: ["A1", "B1"], + multi: false, + selectedOptions: ["A1"], + customInput: undefined, + note: undefined, + timedOut: undefined, + }, + { + id: "q2", + question: "Q2?", + options: ["A2", "B2"], + multi: false, + selectedOptions: ["B2"], + customInput: undefined, + note: undefined, + timedOut: undefined, + }, + ]); + }); + + it("multi-select: Space and Enter toggle without advancing, Next row advances", () => { + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const onChat = vi.fn(); + const onPrompt = vi.fn(); + + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose multiple?", + options: [{ label: "Option A" }, { label: "Option B" }], + multi: true, + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat, + onPrompt, + }); + + // Space on Option A - toggles A + component.handleInput(SPACE); + + // Down to Option B, Enter - toggles B + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onSubmit).not.toHaveBeenCalled(); + + // Down to Other + component.handleInput(DOWN); + // Down to Next + component.handleInput(DOWN); + // Enter on Next to submit + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option A", "Option B"]); + }); + + it("tab-state persistence: answer question 0, Tab forward, Tab back, answer still present", () => { + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const onChat = vi.fn(); + const onPrompt = vi.fn(); + + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Q1?", + options: [{ label: "A1" }, { label: "B1" }], + }, + { + id: "q2", + question: "Q2?", + options: [{ label: "A2" }, { label: "B2" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat, + onPrompt, + }); + + // Enter on A1 selects it and auto-advances to Q2 (tab 1) + component.handleInput(ENTER); + + // Shift+Tab back to Q1 (tab 0) + component.handleInput(SHIFT_TAB); + + // Enter again on Q1's currently selected option (which will re-select/keep it and auto-advance to Q2) + component.handleInput(ENTER); + + // On Q2: select B2 and advance to Submit + component.handleInput(DOWN); + component.handleInput(ENTER); + + // On Submit: Enter to submit + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["A1"]); + expect(onSubmit.mock.calls[0][0].results[1].selectedOptions).toEqual(["B2"]); + }); + + it("Tab and Shift+Tab switches tabs", () => { + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Q1?", + options: [{ label: "A1" }, { label: "B1" }], + }, + { + id: "q2", + question: "Q2?", + options: [{ label: "A2" }, { label: "B2" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + + // Tab from Q1 -> Q2 + component.handleInput(TAB); + // Tab from Q2 -> Submit + component.handleInput(TAB); + // Shift+Tab from Submit -> Q2 + component.handleInput(SHIFT_TAB); + + // Down to B2, Enter -> Submit + component.handleInput(DOWN); + component.handleInput(ENTER); + + // Enter on Submit + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual([]); + expect(onSubmit.mock.calls[0][0].results[1].selectedOptions).toEqual(["B2"]); + }); + + it("Submit tab shows unanswered warning but Enter still submits", () => { + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Q1?", + options: [{ label: "A1" }, { label: "B1" }], + }, + { + id: "q2", + question: "Q2?", + options: [{ label: "A2" }, { label: "B2" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + + // Tab to Submit + component.handleInput(TAB); + component.handleInput(TAB); + + const output = render(component); + expect(output.toLowerCase()).toContain("unanswered"); + + // Enter on Submit + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual([]); + expect(onSubmit.mock.calls[0][0].results[1].selectedOptions).toEqual([]); + }); + + it("Esc/cancel fires onCancel", () => { + const onCancel = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit: vi.fn(), + onCancel, + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + + component.handleInput(CANCEL); + expect(onCancel).toHaveBeenCalledTimes(1); + }); + + it("selecting 'Chat about this' on a question tab fires onChat", () => { + const onChat = vi.fn(); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Q1?", + options: [{ label: "A1" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat, + onPrompt: vi.fn(), + }); + + // Cursor positions: + // 0: A1 + // 1: Other + // 2: Chat about this + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onChat).toHaveBeenCalledTimes(1); + expect(onSubmit).not.toHaveBeenCalled(); + }); + + it("selecting 'Chat about this' on Submit tab fires onChat", () => { + const onChat = vi.fn(); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Q1?", + options: [{ label: "A1" }], + }, + { + id: "q2", + question: "Q2?", + options: [{ label: "A2" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat, + onPrompt: vi.fn(), + }); + + // Tab to Submit + component.handleInput(TAB); + component.handleInput(TAB); + + // Cursor positions on Submit tab: + // 0: Submit + // 1: Chat about this + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onChat).toHaveBeenCalledTimes(1); + expect(onSubmit).not.toHaveBeenCalled(); + }); + + it("n on an option calls onPrompt and stores note with marker", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("My Custom Note")); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + // Highlight is on Option A. Press 'n'. + component.handleInput("n"); + + // Await microtasks so the async #promptForNote runs + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(1); + expect(onPrompt.mock.calls[0][0]).toBe("Note for Option A: Choose one?"); + + // Verify note is saved by submitting + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].note).toBe("My Custom Note"); + }); + + it("omits a note when a single-select answer changes to a different option", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("Note for A")); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option B"]); + expect(onSubmit.mock.calls[0][0].results[0].note).toBeUndefined(); + }); + + it("clears the note when a noted multi-select option is toggled off", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("Note for A")); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose multiple?", + options: [{ label: "Option A" }, { label: "Option B" }], + multi: true, + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + component.handleInput(SPACE); + component.handleInput(SPACE); + expect(render(component)).not.toContain("✎ note"); + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual([]); + expect(onSubmit.mock.calls[0][0].results[0].note).toBeUndefined(); + }); + + it("shows selected multi-select options together with custom input on Submit", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("custom detail")); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose multiple?", + options: [{ label: "Option A" }, { label: "Option B" }], + multi: true, + }, + { + id: "q2", + question: "Second question?", + options: [{ label: "Option C" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + component.handleInput(SPACE); + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + await Promise.resolve(); + await Promise.resolve(); + + component.handleInput(TAB); + const review = render(component); + expect(review).toContain("Option A"); + expect(review).toContain("custom detail"); + + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option A"]); + expect(onSubmit.mock.calls[0][0].results[0].customInput).toBe("custom detail"); + }); + + it("defers a timeout that fires during a pending prompt and honors the resolved custom input", async () => { + vi.useFakeTimers(); + const deferred = Promise.withResolvers(); + const onPrompt = vi.fn().mockReturnValue(deferred.promise); + const onSubmit = vi.fn(); + const onTimeout = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "First?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + { + id: "q2", + question: "Second?", + options: [{ label: "Option C" }, { label: "Option D" }], + recommended: 1, + }, + ]; + + const component = new AskDialogComponent( + questions, + { onSubmit, onCancel: vi.fn(), onChat: vi.fn(), onPrompt }, + { timeout: 1000, onTimeout }, + ); + + // Open the "Other (type your own)" prompt on question 1. + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + expect(onPrompt).toHaveBeenCalledTimes(1); + + // Timer expires while the prompt is pending: the timeout must be deferred, + // not submit the recommended fallback out from under the user. + vi.advanceTimersByTime(1000); + expect(onTimeout).not.toHaveBeenCalled(); + expect(onSubmit).not.toHaveBeenCalled(); + + // Resolving the prompt honors the typed answer, then runs the deferred + // timeout handling exactly once. + deferred.resolve("my answer"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onTimeout).toHaveBeenCalledTimes(1); + expect(onSubmit).toHaveBeenCalledTimes(1); + const results = onSubmit.mock.calls[0][0].results; + expect(results[0].customInput).toBe("my answer"); + expect(results[0].selectedOptions).toEqual([]); + expect(results[0].timedOut).toBeUndefined(); + expect(results[1].selectedOptions).toEqual(["Option D"]); + expect(results[1].timedOut).toBe(true); + }); + + it("keeps a single-question custom prompt answer when timeout expires while the prompt is pending", async () => { + vi.useFakeTimers(); + const deferred = Promise.withResolvers(); + const onPrompt = vi.fn().mockReturnValue(deferred.promise); + const onSubmit = vi.fn(); + const onTimeout = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Only question?", + options: [{ label: "Fallback" }], + }, + ]; + + const component = new AskDialogComponent( + questions, + { onSubmit, onCancel: vi.fn(), onChat: vi.fn(), onPrompt }, + { timeout: 1000, onTimeout }, + ); + + component.handleInput(DOWN); + component.handleInput(ENTER); + expect(onPrompt).toHaveBeenCalledTimes(1); + + vi.advanceTimersByTime(1000); + expect(onTimeout).not.toHaveBeenCalled(); + expect(onSubmit).not.toHaveBeenCalled(); + + deferred.resolve("my answer"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onTimeout).not.toHaveBeenCalled(); + const result = onSubmit.mock.calls[0][0].results[0]; + expect(result.customInput).toBe("my answer"); + expect(result.selectedOptions).toEqual([]); + expect(result.timedOut).toBeUndefined(); + }); + + it("uses a noted non-recommended option as the timeout fallback", async () => { + vi.useFakeTimers(); + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("why B")); + const onSubmit = vi.fn(); + const onTimeout = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + recommended: 0, + }, + ]; + + const component = new AskDialogComponent( + questions, + { onSubmit, onCancel: vi.fn(), onChat: vi.fn(), onPrompt }, + { timeout: 1000, onTimeout }, + ); + + component.handleInput(DOWN); + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + vi.advanceTimersByTime(1000); + + expect(onTimeout).toHaveBeenCalledTimes(1); + expect(onSubmit).toHaveBeenCalledTimes(1); + const result = onSubmit.mock.calls[0][0].results[0]; + expect(result.selectedOptions).toEqual(["Option B"]); + expect(result.note).toBe("why B"); + expect(result.timedOut).toBe(true); + }); + + it("preserves a pending note on a non-recommended option when deferred timeout submits", async () => { + vi.useFakeTimers(); + const deferred = Promise.withResolvers(); + const onPrompt = vi.fn().mockReturnValue(deferred.promise); + const onSubmit = vi.fn(); + const onTimeout = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + recommended: 0, + }, + ]; + + const component = new AskDialogComponent( + questions, + { onSubmit, onCancel: vi.fn(), onChat: vi.fn(), onPrompt }, + { timeout: 1000, onTimeout }, + ); + + component.handleInput(DOWN); + component.handleInput("n"); + expect(onPrompt).toHaveBeenCalledTimes(1); + + vi.advanceTimersByTime(1000); + expect(onTimeout).not.toHaveBeenCalled(); + expect(onSubmit).not.toHaveBeenCalled(); + + deferred.resolve("why B"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onTimeout).toHaveBeenCalledTimes(1); + expect(onSubmit).toHaveBeenCalledTimes(1); + const result = onSubmit.mock.calls[0][0].results[0]; + expect(result.selectedOptions).toEqual(["Option B"]); + expect(result.note).toBe("why B"); + expect(result.timedOut).toBe(true); + }); +}); diff --git a/packages/coding-agent/test/modes/components/settings-layout.test.ts b/packages/coding-agent/test/modes/components/settings-layout.test.ts index dbcbd1b0e..88f89b798 100644 --- a/packages/coding-agent/test/modes/components/settings-layout.test.ts +++ b/packages/coding-agent/test/modes/components/settings-layout.test.ts @@ -97,4 +97,14 @@ describe("settings layout", () => { group: "Services", }); }); + + it("exposes ask.enabled as a boolean under Available Tools", () => { + const def = getSettingsForTab("tools").find(def => def.path === "ask.enabled"); + + expect(def).toMatchObject({ + type: "boolean", + label: "Ask", + group: "Available Tools", + }); + }); }); diff --git a/packages/coding-agent/test/tools/ask.test.ts b/packages/coding-agent/test/tools/ask.test.ts index efd7b3c30..ed2b68bdf 100644 --- a/packages/coding-agent/test/tools/ask.test.ts +++ b/packages/coding-agent/test/tools/ask.test.ts @@ -2,11 +2,16 @@ import { beforeAll, describe, expect, it, vi } from "bun:test"; import { stripVTControlCharacters } from "node:util"; import type { AgentToolContext } from "@oh-my-pi/pi-agent-core"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import type { ExtensionUISelectItem } from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; +import type { + ExtensionAskDialogQuestion, + ExtensionAskDialogResult, + ExtensionUISelectItem, +} from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; import { getThemeByName, initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { AskTool, askToolRenderer } from "@oh-my-pi/pi-coding-agent/tools/ask"; import { ToolAbortError } from "@oh-my-pi/pi-coding-agent/tools/tool-errors"; +import { type } from "arktype"; function createSession(overrides: Partial = {}): ToolSession { return { @@ -20,7 +25,7 @@ function createSession(overrides: Partial = {}): ToolSession { } function createContext(args: { - select: ( + select?: ( prompt: string, options: ExtensionUISelectItem[], dialogOptions?: { @@ -42,13 +47,18 @@ function createContext(args: { dialogOptions?: { signal?: AbortSignal }, editorOptions?: { promptStyle?: boolean }, ) => Promise; + askDialog?: ( + questions: ExtensionAskDialogQuestion[], + dialogOptions?: any, + ) => Promise; abort?: () => void; }): AgentToolContext { // AgentToolContext includes many runtime fields; tests only need UI + abort behavior. return { hasUI: true, ui: { - select: args.select, + ...(args.select ? { select: args.select } : {}), + ...(args.askDialog ? { askDialog: args.askDialog } : {}), editor: ( title: string, prefill?: string, @@ -1528,3 +1538,135 @@ describe("askToolRenderer malformed call args", () => { expect(text).toContain("Proper"); }); }); + +describe("AskTool rich ask dialog", () => { + it("accepts new schema fields (header, preview, note) and maps them into AskToolDetails", async () => { + const tool = new AskTool(createSession()); + const askDialog = vi.fn().mockResolvedValue({ + kind: "submit", + results: [ + { + id: "q1", + question: "Q1?", + options: ["Option A"], + multi: false, + selectedOptions: ["Option A"], + note: "My Custom Note", + timedOut: undefined, + }, + ], + }); + const context = createContext({ askDialog }); + + const result = await tool.execute( + "call-rich-dialog", + { + questions: [ + { + id: "q1", + question: "Q1?", + header: "Chip Header", + options: [{ label: "Option A", preview: "My Preview" }], + }, + ], + }, + undefined, + undefined, + context, + ); + + expect(askDialog).toHaveBeenCalledTimes(1); + // Check that header and preview were forwarded + expect(askDialog.mock.calls[0][0]).toEqual([ + { + id: "q1", + question: "Q1?", + header: "Chip Header", + options: [{ label: "Option A", preview: "My Preview" }], + }, + ]); + + // Verify result contains details with note mapping + expect(result.details).toEqual({ + question: "Q1?", + options: ["Option A"], + multi: false, + selectedOptions: ["Option A"], + customInput: undefined, + note: "My Custom Note", + timedOut: undefined, + }); + }); + + it("aborts and throws ToolAbortError when askDialog returns undefined", async () => { + const tool = new AskTool(createSession()); + const abort = vi.fn(); + const askDialog = vi.fn().mockResolvedValue(undefined); + const context = createContext({ askDialog, abort }); + + await expect( + tool.execute( + "call-rich-dialog-cancel", + { + questions: [{ id: "q1", question: "Q1?", options: [{ label: "Option A" }] }], + }, + undefined, + undefined, + context, + ), + ).rejects.toThrow(ToolAbortError); + + expect(abort).toHaveBeenCalledTimes(1); + }); + + it("ignores preview and header in degraded select path", async () => { + const tool = new AskTool(createSession()); + const select = vi.fn().mockResolvedValue("Option A"); + const context = createContext({ select }); + + await tool.execute( + "call-degraded", + { + questions: [ + { + id: "q1", + question: "Q1?", + header: "Chip Header", + options: [{ label: "Option A", description: "Desc A", preview: "My Preview" }], + }, + ], + }, + undefined, + undefined, + context, + ); + + expect(select).toHaveBeenCalledTimes(1); + // verify preview/header are NOT forwarded to select options + expect(select.mock.calls[0][1]).toEqual([{ label: "Option A", description: "Desc A" }, "Other (type your own)"]); + }); + + it("rejects reserved-label collision in parameters validation", async () => { + const tool = new AskTool(createSession()); + + const valid = tool.parameters({ + questions: [{ id: "q1", question: "Q?", options: [{ label: "ok" }] }], + }); + expect(valid instanceof type.errors).toBe(false); + + const reservedOther = tool.parameters({ + questions: [{ id: "q1", question: "Q?", options: [{ label: "Other (type your own)" }] }], + }); + expect(reservedOther instanceof type.errors).toBe(true); + + const reservedChat = tool.parameters({ + questions: [{ id: "q1", question: "Q?", options: [{ label: "Chat about this" }] }], + }); + expect(reservedChat instanceof type.errors).toBe(true); + + const reservedNext = tool.parameters({ + questions: [{ id: "q1", question: "Q?", options: [{ label: "Next →" }] }], + }); + expect(reservedNext instanceof type.errors).toBe(true); + }); +}); diff --git a/packages/coding-agent/test/tools/index.test.ts b/packages/coding-agent/test/tools/index.test.ts index eae16b79c..09a2b3aff 100644 --- a/packages/coding-agent/test/tools/index.test.ts +++ b/packages/coding-agent/test/tools/index.test.ts @@ -208,6 +208,27 @@ describe("createTools", () => { expect(names).toContain("ask"); }); + it("excludes ask tool when ask.enabled is false", async () => { + const session = createTestSession({ + hasUI: true, + settings: createSettingsWithOverrides({ "ask.enabled": false }), + }); + const tools = await createTools(session); + expect(tools.map(t => t.name)).not.toContain("ask"); + + const requested = await createTools(session, ["ask", "read"]); + expect(requested.map(t => t.name)).toEqual(["read", "resolve"]); + }); + + it("includes ask tool when ask.enabled is true and hasUI is true", async () => { + const session = createTestSession({ + hasUI: true, + settings: createSettingsWithOverrides({ "ask.enabled": true }), + }); + const tools = await createTools(session); + expect(tools.map(t => t.name)).toContain("ask"); + }); + it("filters disabled builtin tools by settings", async () => { const session = createTestSession({ settings: createSettingsWithOverrides({ From 3c2c9f5bca3a0264f3a80aa2672ed286300bc50f Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Fri, 3 Jul 2026 09:21:40 +0900 Subject: [PATCH 002/205] docs(coding-agent): update changelog for u10 Adds the issue-linked Unreleased changelog entry for this publish branch. Op: extend --- packages/coding-agent/CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 220bf8ad0..7b6d9651d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added rich interactive ask dialogs with headers, previews, notes, chat redirect, and ask.enabled gating ([#4186](https://github.com/can1357/oh-my-pi/issues/4186)). + ## [16.3.3] - 2026-07-02 ### Breaking Changes From 69c02c802a2861cec59b133094aecebb9cf4c6e9 Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Fri, 3 Jul 2026 10:16:41 +0900 Subject: [PATCH 003/205] fix(ask): reset countdown on input, bound prompt titles, fix scroll viewport - Reset the inactivity countdown in handleInput after the closed/prompt guard, matching HookSelector/HookInput semantics so a user actively navigating options/tabs is not auto-submitted by an absolute deadline. - Add boundPromptTitle helper that flattens whitespace, wraps to the terminal content width, and caps at 3 rows with ellipsis truncation; apply it to custom-input and note prompt titles in the rich dialog and the guest-UI editor path so long/multi-line questions stay usable. - Drop totalRows from both ScrollView calls so the full allLines array is sliced via scrollOffset + row; previously totalRows caused render to read lines[row] instead of lines[scrollOffset + row], leaving the viewport stuck at the top while the scrollbar thumb moved. - Add tests for inactivity reset, bounded prompt titles, and scrolling. Restores: review:4375 Op: correct --- .../src/modes/components/ask-dialog.ts | 39 +++- .../controllers/extension-ui-controller.ts | 6 +- .../test/modes/components/ask-dialog.test.ts | 170 ++++++++++++++++++ 3 files changed, 208 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/modes/components/ask-dialog.ts b/packages/coding-agent/src/modes/components/ask-dialog.ts index de96dcef3..979871f6d 100644 --- a/packages/coding-agent/src/modes/components/ask-dialog.ts +++ b/packages/coding-agent/src/modes/components/ask-dialog.ts @@ -39,6 +39,30 @@ const SIDE_BY_SIDE_LIST_MIN_WIDTH = 30; const SIDE_BY_SIDE_GAP_WIDTH = 3; const MAX_HEADER_CHIP_WIDTH = 16; const PREVIEW_HEADER = "Preview"; +/** Maximum number of title lines shown in the prompt editor overlay, so a + * long or multi-line question cannot push the input row off-screen. Mirrors + * the bounded-title pattern from the legacy ask path without its option-window + * coupling. */ +const MAX_PROMPT_TITLE_ROWS = 3; +/** Border (2) + padX (2) columns consumed by the HookEditor chrome. */ +const PROMPT_TITLE_CHROME_COLUMNS = 4; + +function promptTitleContentWidth(): number { + const cols = process.stdout.columns ?? 80; + return Math.max(1, cols - PROMPT_TITLE_CHROME_COLUMNS); +} + +/** Bound a prompt editor title to a fixed row/width budget so long or + * multi-line questions stay usable inside the small prompt overlay. */ +export function boundPromptTitle(prefix: string, question: string): string { + const width = promptTitleContentWidth(); + const flat = normalizedInlineInput(`${prefix}${question}`); + const wrapped = wrapTextWithAnsi(flat, width); + if (wrapped.length <= MAX_PROMPT_TITLE_ROWS) return wrapped.join("\n"); + const kept = wrapped.slice(0, MAX_PROMPT_TITLE_ROWS - 1); + const last = truncateToWidth(wrapped[MAX_PROMPT_TITLE_ROWS - 1] ?? "", width, Ellipsis.Unicode); + return [...kept, last].join("\n"); +} interface AskDialogCallbacks { onSubmit(result: ExtensionAskDialogResult): void; @@ -324,6 +348,9 @@ export class AskDialogComponent implements Component { handleInput(keyData: string): void { if (this.#closed || this.#promptActive) return; + // Reset the inactivity countdown on any key that reaches past the + // closed/prompt guards, matching HookSelector/HookInput semantics. + this.#countdown?.reset(); if (matchesSelectCancel(keyData)) { this.#finishCancel(); return; @@ -549,7 +576,10 @@ export class AskDialogComponent implements Component { ): Promise { this.#promptActive = true; try { - const input = await this.callbacks.onPrompt(`Custom answer: ${question.question}`, state.customInput); + const input = await this.callbacks.onPrompt( + boundPromptTitle("Custom answer: ", question.question), + state.customInput, + ); if (input === undefined || this.#closed) return; state.customInput = input; if (!question.multi) { @@ -571,7 +601,10 @@ export class AskDialogComponent implements Component { ): Promise { this.#promptActive = true; try { - const input = await this.callbacks.onPrompt(`Note for ${rowItem.label}: ${question.question}`, state.note); + const input = await this.callbacks.onPrompt( + boundPromptTitle(`Note for ${rowItem.label}: `, question.question), + state.note, + ); if (input === undefined || this.#closed) return; state.note = input; state.noteRowKey = rowItem.key; @@ -636,7 +669,6 @@ export class AskDialogComponent implements Component { const scrollView = new ScrollView(allLines, { height: rows, scrollbar: "auto", - totalRows: allLines.length, theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) }, }); scrollView.setScrollOffset(state.scrollOffset); @@ -710,7 +742,6 @@ export class AskDialogComponent implements Component { const scrollView = new ScrollView(allLines, { height: rows, scrollbar: "auto", - totalRows: allLines.length, theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) }, }); scrollView.setScrollOffset(this.#submitScrollOffset); diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index 3fd1b5723..aefb6ea6a 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -22,7 +22,7 @@ import type { } from "../../extensibility/extensions"; import { getSessionSlashCommands } from "../../extensibility/extensions/get-commands-handler"; import { createExtensionModelQuery } from "../../extensibility/extensions/model-api"; -import { AskDialogComponent } from "../../modes/components/ask-dialog"; +import { AskDialogComponent, boundPromptTitle } from "../../modes/components/ask-dialog"; import { HookEditorComponent } from "../../modes/components/hook-editor"; import { HookInputComponent } from "../../modes/components/hook-input"; import { HookSelectorComponent, type HookSelectorSlider } from "../../modes/components/hook-selector"; @@ -770,7 +770,7 @@ export class ExtensionUiController { if (choice === ASK_NEXT_OPTION) break; if (choice === ASK_OTHER_OPTION) { const input = await this.#requestGuestUiString( - { kind: "editor", title: `Custom answer: ${question.question}` }, + { kind: "editor", title: boundPromptTitle("Custom answer: ", question.question) }, signal, ); if (input === "unavailable" || input === undefined) return input; @@ -802,7 +802,7 @@ export class ExtensionUiController { if (choice === ASK_CHAT_OPTION) return undefined; if (choice === ASK_OTHER_OPTION) { const input = await this.#requestGuestUiString( - { kind: "editor", title: `Custom answer: ${question.question}` }, + { kind: "editor", title: boundPromptTitle("Custom answer: ", question.question) }, signal, ); if (input === "unavailable" || input === undefined) return input; diff --git a/packages/coding-agent/test/modes/components/ask-dialog.test.ts b/packages/coding-agent/test/modes/components/ask-dialog.test.ts index f08b67268..88e5e9f6f 100644 --- a/packages/coding-agent/test/modes/components/ask-dialog.test.ts +++ b/packages/coding-agent/test/modes/components/ask-dialog.test.ts @@ -735,4 +735,174 @@ describe("AskDialogComponent", () => { expect(result.note).toBe("why B"); expect(result.timedOut).toBe(true); }); + + it("resets the inactivity countdown on user input after the closed/prompt guard", () => { + vi.useFakeTimers(); + const onTimeout = vi.fn(); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent( + questions, + { onSubmit, onCancel: vi.fn(), onChat: vi.fn(), onPrompt: vi.fn() }, + { timeout: 5000, onTimeout }, + ); + + // Advance most of the timeout window. + vi.advanceTimersByTime(4000); + expect(onTimeout).not.toHaveBeenCalled(); + + // User input (DOWN) should reset the countdown. + component.handleInput(DOWN); + + // Advancing past the *original* deadline must NOT fire the timeout — + // the reset moved the deadline forward by the interaction. + vi.advanceTimersByTime(2000); + expect(onTimeout).not.toHaveBeenCalled(); + + // Advancing the remaining time after the reset DOES fire. + vi.advanceTimersByTime(3000); + expect(onTimeout).toHaveBeenCalledTimes(1); + }); + + it("does not reset the countdown while a prompt is active", async () => { + vi.useFakeTimers(); + const deferred = Promise.withResolvers(); + const onPrompt = vi.fn().mockReturnValue(deferred.promise); + const onTimeout = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }], + }, + ]; + + const component = new AskDialogComponent( + questions, + { onSubmit: vi.fn(), onCancel: vi.fn(), onChat: vi.fn(), onPrompt }, + { timeout: 5000, onTimeout }, + ); + + // Open the custom-input prompt (DOWN to "Other", ENTER). + component.handleInput(DOWN); + component.handleInput(ENTER); + expect(onPrompt).toHaveBeenCalledTimes(1); + + // While the prompt is pending, input is guarded — no reset. + component.handleInput(DOWN); + vi.advanceTimersByTime(5000); + // Timeout is deferred during prompt, not fired. + expect(onTimeout).not.toHaveBeenCalled(); + + deferred.resolve("answer"); + await Promise.resolve(); + await Promise.resolve(); + }); + + it("bounds custom input prompt title for long multi-line questions", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("custom")); + const longQuestion = "This is a very long question ".repeat(20); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: longQuestion, + options: [{ label: "Option A" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit: vi.fn(), + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + // Navigate to "Other" and press Enter to trigger the custom prompt. + component.handleInput(DOWN); + component.handleInput(ENTER); + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(1); + const title = onPrompt.mock.calls[0][0] as string; + const lines = title.split("\n"); + // Title must be bounded to at most MAX_PROMPT_TITLE_ROWS lines. + expect(lines.length).toBeLessThanOrEqual(3); + // Each line must fit within the terminal content width. + for (const line of lines) { + expect(stripVTControlCharacters(line).length).toBeLessThanOrEqual((process.stdout.columns ?? 80) - 4); + } + // Must contain the prefix and a truncation indicator on the last line. + expect(stripVTControlCharacters(title)).toContain("Custom answer:"); + }); + + it("bounds note prompt title for long multi-line questions", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve("note")); + const longQuestion = "Multi\nline\nquestion ".repeat(30); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: longQuestion, + options: [{ label: "Option A" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit: vi.fn(), + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + // Press 'n' on the highlighted option to trigger the note prompt. + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(1); + const title = onPrompt.mock.calls[0][0] as string; + const lines = title.split("\n"); + // Title must be bounded to at most MAX_PROMPT_TITLE_ROWS lines. + expect(lines.length).toBeLessThanOrEqual(3); + // The multi-line question must be flattened (no raw newlines expanding rows). + expect(stripVTControlCharacters(title)).toContain("Note for Option A:"); + }); + + it("scrolls question rows when cursor moves below the viewport", () => { + // Use many options so the rendered list overflows a small body. + const options = Array.from({ length: 30 }, (_, i) => ({ label: `Option ${String(i + 1).padStart(2, "0")}` })); + const questions: ExtensionAskDialogQuestion[] = [{ id: "q1", question: "Pick one?", options }]; + + const component = new AskDialogComponent(questions, { + onSubmit: vi.fn(), + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + + // Render at a narrow width / small height to force overflow. + // The body height is derived from process.stdout.rows; we render at + // width 60 and inspect the visible content. + const renderAt = (width: number): string => stripVTControlCharacters(component.render(width).join("\n")); + + // Initial render: first options visible, last options not. + const initial = renderAt(60); + expect(initial).toContain("Option 01"); + expect(initial).not.toContain("Option 30"); + + // Move cursor down past the viewport boundary to trigger scrolling. + for (let i = 0; i < 28; i++) component.handleInput(DOWN); + + const scrolled = renderAt(60); + // After scrolling, early options should be gone and later ones visible. + expect(scrolled).not.toContain("Option 01"); + expect(scrolled).toContain("Option 29"); + }); }); From 48b2a742c8520cec275fae505a0947c4320d7ee6 Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Fri, 3 Jul 2026 11:52:51 +0900 Subject: [PATCH 004/205] fix(ask): distinct chat redirect result, row-specific note prefill Op: correct Restores: review:4375 - Widen ExtensionAskDialogResult to a union with { kind: "chat" } variant so AskTool can distinguish chat handoff from cancel (undefined). - AskDialogComponent.#finishChat passes { kind: "chat" } via onChat. - Controller settles { kind: "chat" } locally and propagates a "chat" sentinel through the guest/collab path instead of returning undefined. - AskTool returns a chat-redirect AgentToolResult (chatRedirect details) instead of aborting with ToolAbortError. - #promptForNote prefills with the existing note only when editing the same row (noteRowKey === rowItem.key), preventing cross-row note leaks. - Add ask-dialog and ask tool tests for chat redirect and row-specific note prefill. --- .../src/extensibility/extensions/types.ts | 11 ++- .../src/modes/components/ask-dialog.ts | 11 +-- .../controllers/extension-ui-controller.ts | 9 +- packages/coding-agent/src/tools/ask.ts | 29 ++++++ .../test/modes/components/ask-dialog.test.ts | 89 ++++++++++++++++++- packages/coding-agent/test/tools/ask.test.ts | 22 +++++ 6 files changed, 160 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 03e3da10e..897f88f82 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -139,11 +139,20 @@ export interface ExtensionAskDialogResultItem { timedOut?: boolean; } -export interface ExtensionAskDialogResult { +export interface ExtensionAskDialogSubmitResult { kind: "submit"; results: ExtensionAskDialogResultItem[]; } +/** Chat-redirect result: the user chose "Chat about this" instead of + * answering. Distinct from `undefined` (cancel) so AskTool can hand off to + * the chat loop rather than aborting. */ +export interface ExtensionAskDialogChatResult { + kind: "chat"; +} + +export type ExtensionAskDialogResult = ExtensionAskDialogSubmitResult | ExtensionAskDialogChatResult; + export function getExtensionUISelectOptionLabel(option: ExtensionUISelectItem): string { return typeof option === "string" ? option : option.label; } diff --git a/packages/coding-agent/src/modes/components/ask-dialog.ts b/packages/coding-agent/src/modes/components/ask-dialog.ts index 979871f6d..3fc43eaca 100644 --- a/packages/coding-agent/src/modes/components/ask-dialog.ts +++ b/packages/coding-agent/src/modes/components/ask-dialog.ts @@ -17,9 +17,10 @@ import { wrapTextWithAnsi, } from "@oh-my-pi/pi-tui"; import type { + ExtensionAskDialogChatResult, ExtensionAskDialogQuestion, - ExtensionAskDialogResult, ExtensionAskDialogResultItem, + ExtensionAskDialogSubmitResult, } from "../../extensibility/extensions"; import { getTabBarTheme } from "../shared"; import { getMarkdownTheme, highlightCode, theme } from "../theme/theme"; @@ -65,9 +66,9 @@ export function boundPromptTitle(prefix: string, question: string): string { } interface AskDialogCallbacks { - onSubmit(result: ExtensionAskDialogResult): void; + onSubmit(result: ExtensionAskDialogSubmitResult): void; onCancel(): void; - onChat(): void; + onChat(result: ExtensionAskDialogChatResult): void; onPrompt(title: string, prefill?: string): Promise; } @@ -603,7 +604,7 @@ export class AskDialogComponent implements Component { try { const input = await this.callbacks.onPrompt( boundPromptTitle(`Note for ${rowItem.label}: `, question.question), - state.note, + state.noteRowKey === rowItem.key ? state.note : undefined, ); if (input === undefined || this.#closed) return; state.note = input; @@ -840,7 +841,7 @@ export class AskDialogComponent implements Component { if (this.#closed) return; this.#closed = true; this.#countdown?.dispose(); - this.callbacks.onChat(); + this.callbacks.onChat({ kind: "chat" }); } #buildResults(): ExtensionAskDialogResultItem[] { diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index aefb6ea6a..b512aab64 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -665,7 +665,7 @@ export class ExtensionUiController { { onSubmit: result => settle(result), onCancel: () => settle(undefined), - onChat: () => settle(undefined), + onChat: () => settle({ kind: "chat" }), onPrompt: promptForText, }, { @@ -734,6 +734,7 @@ export class ExtensionUiController { for (const question of questions) { const result = await this.#runGuestAskQuestion(question, signal); if (result === "unavailable" || result === undefined) return result; + if (result === "chat") return { kind: "chat" }; results.push(result); } return { kind: "submit", results }; @@ -742,7 +743,7 @@ export class ExtensionUiController { async #runGuestAskQuestion( question: ExtensionAskDialogQuestion, signal: AbortSignal, - ): Promise { + ): Promise { const selected = new Set(); let customInput: string | undefined; const baseOptions: CollabUiSelectItem[] = question.options.map(option => @@ -766,7 +767,7 @@ export class ExtensionUiController { signal, ); if (choice === "unavailable" || choice === undefined) return choice; - if (choice === ASK_CHAT_OPTION) return undefined; + if (choice === ASK_CHAT_OPTION) return "chat"; if (choice === ASK_NEXT_OPTION) break; if (choice === ASK_OTHER_OPTION) { const input = await this.#requestGuestUiString( @@ -799,7 +800,7 @@ export class ExtensionUiController { signal, ); if (choice === "unavailable" || choice === undefined) return choice; - if (choice === ASK_CHAT_OPTION) return undefined; + if (choice === ASK_CHAT_OPTION) return "chat"; if (choice === ASK_OTHER_OPTION) { const input = await this.#requestGuestUiString( { kind: "editor", title: boundPromptTitle("Custom answer: ", question.question) }, diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index d6663962c..7cbbe40e5 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -107,6 +107,10 @@ export interface AskToolDetails { timedOut?: boolean; /** Multi-part question mode */ results?: QuestionResult[]; + /** Chat redirect: the user chose "Chat about this" instead of answering. */ + chatRedirect?: boolean; + /** Questions surfaced when chatRedirect is true. */ + questions?: string[]; } interface AskOption { @@ -857,6 +861,18 @@ export class AskTool implements AgentTool { context.abort(); throw new ToolAbortError("Ask tool was cancelled by the user"); } + if (richResult.kind === "chat") { + const questionText = params.questions.map(q => q.question).join("\n"); + return { + content: [ + { + type: "text" as const, + text: `User chose to chat about this instead of answering.\n\nQuestions asked:\n${questionText}`, + }, + ], + details: { chatRedirect: true, questions: params.questions.map(q => q.question) }, + }; + } if (richResult.results.length !== params.questions.length) { throw new Error("Ask dialog returned a result count that does not match the requested questions"); } @@ -1286,6 +1302,19 @@ export const askToolRenderer = { return new Text(`${header}${body}`, 0, 0); } + // Chat redirect: user chose "Chat about this" instead of answering. + if (details.chatRedirect) { + const header = renderStatusLine({ icon: "info", title: "Ask", meta: ["chat redirect"] }, uiTheme); + const questions = details.questions ?? []; + return framedBlock(uiTheme, width => ({ + header, + sections: questions.length > 0 ? [{ lines: questions.flatMap(q => md(q, width)) }] : [], + state: "warning", + borderColor: "borderMuted", + width, + })); + } + // Multi-part results: one divider-labelled section per question. if (details.results && details.results.length > 0) { const results = details.results; diff --git a/packages/coding-agent/test/modes/components/ask-dialog.test.ts b/packages/coding-agent/test/modes/components/ask-dialog.test.ts index 88e5e9f6f..bd2d734c4 100644 --- a/packages/coding-agent/test/modes/components/ask-dialog.test.ts +++ b/packages/coding-agent/test/modes/components/ask-dialog.test.ts @@ -382,6 +382,7 @@ describe("AskDialogComponent", () => { component.handleInput(ENTER); expect(onChat).toHaveBeenCalledTimes(1); + expect(onChat.mock.calls[0][0]).toEqual({ kind: "chat" }); expect(onSubmit).not.toHaveBeenCalled(); }); @@ -417,8 +418,8 @@ describe("AskDialogComponent", () => { // 1: Chat about this component.handleInput(DOWN); component.handleInput(ENTER); - expect(onChat).toHaveBeenCalledTimes(1); + expect(onChat.mock.calls[0][0]).toEqual({ kind: "chat" }); expect(onSubmit).not.toHaveBeenCalled(); }); @@ -457,6 +458,92 @@ describe("AskDialogComponent", () => { expect(onSubmit.mock.calls[0][0].results[0].note).toBe("My Custom Note"); }); + it("note prefill is empty when editing a different row after noting another option", async () => { + const onPrompt = vi.fn(); + const onSubmit = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + // Cursor starts on Option A. Add a note for A. + onPrompt.mockReturnValueOnce(Promise.resolve("Note for A")); + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(1); + expect(onPrompt.mock.calls[0][0]).toBe("Note for Option A: Choose one?"); + // No prior note → prefill is undefined. + expect(onPrompt.mock.calls[0][1]).toBeUndefined(); + + // Move down to Option B and open its note. + component.handleInput(DOWN); + onPrompt.mockReturnValueOnce(Promise.resolve("Note for B")); + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(2); + // Prefill for Option B must be undefined — not the note from Option A. + expect(onPrompt.mock.calls[1][1]).toBeUndefined(); + + // Move back up to Option A and re-open its note. + component.handleInput("\x1b[A"); // UP + onPrompt.mockReturnValueOnce(Promise.resolve("Updated note")); + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(3); + // Note now belongs to Option B, so re-editing Option A starts empty. + expect(onPrompt.mock.calls[2][1]).toBeUndefined(); + }); + + it("note prefill reuses the existing note when re-editing the same row", async () => { + const onPrompt = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit: vi.fn(), + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt, + }); + + // Add a note on Option A. + onPrompt.mockReturnValueOnce(Promise.resolve("My note")); + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + // Re-open the note on the same row (cursor still on Option A). + onPrompt.mockReturnValueOnce(Promise.resolve("Updated note")); + component.handleInput("n"); + await Promise.resolve(); + await Promise.resolve(); + + expect(onPrompt).toHaveBeenCalledTimes(2); + // Same row → prefill reuses the existing note. + expect(onPrompt.mock.calls[1][1]).toBe("My note"); + }); + it("omits a note when a single-select answer changes to a different option", async () => { const onPrompt = vi.fn().mockReturnValue(Promise.resolve("Note for A")); const onSubmit = vi.fn(); diff --git a/packages/coding-agent/test/tools/ask.test.ts b/packages/coding-agent/test/tools/ask.test.ts index ed2b68bdf..0008265f9 100644 --- a/packages/coding-agent/test/tools/ask.test.ts +++ b/packages/coding-agent/test/tools/ask.test.ts @@ -1619,6 +1619,28 @@ describe("AskTool rich ask dialog", () => { expect(abort).toHaveBeenCalledTimes(1); }); + it("returns chat redirect result when askDialog returns kind chat", async () => { + const tool = new AskTool(createSession()); + const abort = vi.fn(); + const askDialog = vi.fn().mockResolvedValue({ kind: "chat" }); + const context = createContext({ askDialog, abort }); + + const result = await tool.execute( + "call-rich-dialog-chat", + { + questions: [{ id: "q1", question: "Q1?", options: [{ label: "Option A" }] }], + }, + undefined, + undefined, + context, + ); + + expect(abort).not.toHaveBeenCalled(); + expect(result.details).toEqual({ chatRedirect: true, questions: ["Q1?"] }); + expect(result.content[0]?.type).toBe("text"); + expect((result.content[0] as { text: string }).text).toContain("chat about this"); + }); + it("ignores preview and header in degraded select path", async () => { const tool = new AskTool(createSession()); const select = vi.fn().mockResolvedValue("Option A"); From 1d14e262bdb94982b87b258257e01336a702c7e5 Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Fri, 3 Jul 2026 13:00:26 +0900 Subject: [PATCH 005/205] fix(ask): tagged guest results, bounded headers, cancel-keeps-open, gated Next - Replace #requestGuestUiString's string|"unavailable"|undefined channel with a tagged GuestUiResult ({answered}|{cancelled}|{unavailable}) so a guest answer literally equal to "unavailable" no longer collides with the transport-unavailable sentinel (PRRT_kwDOQxs0bc6OE3gN). - Cap in-body question header rendering to MAX_HEADER_ROWS with ellipsis truncation so long/multiline questions cannot push options off-screen (PRRT_kwDOQxs0bc6OE3gS). - Guest Other editor cancellation now continues the loop (multi) / re-shows the select (single) instead of cancelling the whole ask (PRRT_kwDOQxs0bc6OE3gU). - Disable the Next row on a single-question multi-select until at least one option or custom input is chosen, preventing empty-result submission (PRRT_kwDOQxs0bc6OE3gY). --- .../src/modes/components/ask-dialog.ts | 34 ++++- .../controllers/extension-ui-controller.ts | 88 +++++++----- .../test/collab/guest-ui-request.test.ts | 42 ++++++ .../test/modes/components/ask-dialog.test.ts | 128 +++++++++++++++++- 4 files changed, 248 insertions(+), 44 deletions(-) diff --git a/packages/coding-agent/src/modes/components/ask-dialog.ts b/packages/coding-agent/src/modes/components/ask-dialog.ts index 3fc43eaca..0b4f5aca0 100644 --- a/packages/coding-agent/src/modes/components/ask-dialog.ts +++ b/packages/coding-agent/src/modes/components/ask-dialog.ts @@ -47,6 +47,10 @@ const PREVIEW_HEADER = "Preview"; const MAX_PROMPT_TITLE_ROWS = 3; /** Border (2) + padX (2) columns consumed by the HookEditor chrome. */ const PROMPT_TITLE_CHROME_COLUMNS = 4; +/** Maximum number of wrapped lines for an in-body question header, so a long + * or multi-line question cannot push the option list off-screen. Mirrors the + * row-cap pattern used by boundPromptTitle for the prompt editor overlay. */ +const MAX_HEADER_ROWS = 4; function promptTitleContentWidth(): number { const cols = process.stdout.columns ?? 80; @@ -96,6 +100,11 @@ interface QuestionRow { key: string; label: string; optionIndex: number | undefined; + /** When true the row is rendered dimmed and Enter/Space on it is ignored. + * Used to gate the Next row on a single-question multi-select until at + * least one option or custom input is chosen, so pressing Next cannot + * submit an empty result. */ + disabled: boolean; } interface SubmitRow { @@ -137,7 +146,14 @@ function renderQuestionTitle(question: ExtensionAskDialogQuestion, index: number const titleWidth = Math.max(1, width - visibleWidth(chip)); const wrapped = wrapTextWithAnsi(questionText, titleWidth); if (wrapped.length === 0) return [chip.trimEnd()]; - return wrapped.map((line, lineIndex) => + const capped = + wrapped.length <= MAX_HEADER_ROWS + ? wrapped + : [ + ...wrapped.slice(0, MAX_HEADER_ROWS - 1), + truncateToWidth(wrapped.slice(MAX_HEADER_ROWS - 1).join(" "), titleWidth, Ellipsis.Unicode), + ]; + return capped.map((line, lineIndex) => lineIndex === 0 ? `${chip}${line}` : `${padding(visibleWidth(chip))}${line}`, ); } @@ -271,10 +287,10 @@ function renderRowLabel( const checked = isOption ? state.selectedOptions.has(stripRecommendedSuffix(rowItem.label)) : isOther && state.customInput !== undefined; - const color = selected ? "accent" : checked ? "toolOutput" : "text"; + const color = rowItem.disabled ? "dim" : selected ? "accent" : checked ? "toolOutput" : "text"; const marker = isOption || isOther ? `${theme.fg(checked ? "success" : "dim", optionMarker(question, checked))} ` : " "; - const cursor = selected ? theme.fg("accent", `${theme.nav.cursor} `) : " "; + const cursor = selected && !rowItem.disabled ? theme.fg("accent", `${theme.nav.cursor} `) : " "; const label = renderInlineMarkdown(rowItem.label, mdTheme, t => theme.fg(color, t)); const noteMarker = state.note && state.noteRowKey === rowItem.key ? theme.fg("success", " ✎ note") : ""; const firstLine = `${cursor}${marker}${label}${noteMarker}`; @@ -450,15 +466,20 @@ export class AskDialogComponent implements Component { } #questionRows(question: ExtensionAskDialogQuestion): QuestionRow[] { + const state = this.#states[this.#currentQuestionIndex()]; + const hasAnswer = !!state && (state.selectedOptions.size > 0 || state.customInput !== undefined); + const nextDisabled = !!question.multi && this.questions.length === 1 && !hasAnswer; const rows: QuestionRow[] = question.options.map((option, index) => ({ kind: "option", key: `option:${index}`, label: this.#optionLabel(question, option.label, index), optionIndex: index, + disabled: false, })); - rows.push({ kind: "other", key: "other", label: OTHER_OPTION, optionIndex: undefined }); - if (question.multi) rows.push({ kind: "next", key: "next", label: NEXT_OPTION, optionIndex: undefined }); - rows.push({ kind: "chat", key: "chat", label: CHAT_ABOUT_THIS_OPTION, optionIndex: undefined }); + rows.push({ kind: "other", key: "other", label: OTHER_OPTION, optionIndex: undefined, disabled: false }); + if (question.multi) + rows.push({ kind: "next", key: "next", label: NEXT_OPTION, optionIndex: undefined, disabled: nextDisabled }); + rows.push({ kind: "chat", key: "chat", label: CHAT_ABOUT_THIS_OPTION, optionIndex: undefined, disabled: false }); return rows; } @@ -504,6 +525,7 @@ export class AskDialogComponent implements Component { return; } if (rowItem.kind === "next") { + if (rowItem.disabled) return; this.#advanceAfterQuestion(); return; } diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index b512aab64..9d812e6a2 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -45,6 +45,12 @@ interface CollabAskDialogWinner { source: "local" | "remote"; value: ExtensionAskDialogResult | undefined; } +/** Tagged result from a guest UI request, distinguishing a real answer (even + * one whose literal value is "unavailable"), an explicit guest cancel, and a + * transport-unavailable sentinel (collab teardown / abort). Replaces the old + * `string | "unavailable" | undefined` channel that let a guest answer of + * "unavailable" collide with the transport sentinel. */ +type GuestUiResult = { kind: "answered"; value: string } | { kind: "cancelled" } | { kind: "unavailable" }; function toWireSelectOptions(options: ExtensionUISelectItem[]): CollabUiSelectItem[] { return options.map(option => @@ -766,20 +772,24 @@ export class ExtensionUiController { }, signal, ); - if (choice === "unavailable" || choice === undefined) return choice; - if (choice === ASK_CHAT_OPTION) return "chat"; - if (choice === ASK_NEXT_OPTION) break; - if (choice === ASK_OTHER_OPTION) { + if (choice.kind === "unavailable") return "unavailable"; + if (choice.kind === "cancelled") return undefined; + if (choice.value === ASK_CHAT_OPTION) return "chat"; + if (choice.value === ASK_NEXT_OPTION) break; + if (choice.value === ASK_OTHER_OPTION) { const input = await this.#requestGuestUiString( { kind: "editor", title: boundPromptTitle("Custom answer: ", question.question) }, signal, ); - if (input === "unavailable" || input === undefined) return input; - customInput = input; + if (input.kind === "unavailable") return "unavailable"; + // Guest cancelled the Other editor: keep the ask open and + // return to the option list instead of cancelling the whole ask. + if (input.kind === "cancelled") continue; + customInput = input.value; break; } - if (selected.has(choice)) selected.delete(choice); - else selected.add(choice); + if (selected.has(choice.value)) selected.delete(choice.value); + else selected.add(choice.value); } } else { const recommended = @@ -787,29 +797,36 @@ export class ExtensionUiController { ? question.recommended : 0; const initialIndex = Math.max(0, Math.min(recommended, Math.max(0, question.options.length - 1))); - const choice = await this.#requestGuestUiString( - { - kind: "select", - title: question.question, - options: [...baseOptions, ASK_OTHER_OPTION, ASK_CHAT_OPTION], - initialIndex, - selectionMarker: "radio", - markableCount: question.options.length, - helpText: "up/down navigate enter select esc cancel", - }, - signal, - ); - if (choice === "unavailable" || choice === undefined) return choice; - if (choice === ASK_CHAT_OPTION) return "chat"; - if (choice === ASK_OTHER_OPTION) { - const input = await this.#requestGuestUiString( - { kind: "editor", title: boundPromptTitle("Custom answer: ", question.question) }, + while (true) { + const choice = await this.#requestGuestUiString( + { + kind: "select", + title: question.question, + options: [...baseOptions, ASK_OTHER_OPTION, ASK_CHAT_OPTION], + initialIndex, + selectionMarker: "radio", + markableCount: question.options.length, + helpText: "up/down navigate enter select esc cancel", + }, signal, ); - if (input === "unavailable" || input === undefined) return input; - customInput = input; - } else { - selected.add(choice); + if (choice.kind === "unavailable") return "unavailable"; + if (choice.kind === "cancelled") return undefined; + if (choice.value === ASK_CHAT_OPTION) return "chat"; + if (choice.value === ASK_OTHER_OPTION) { + const input = await this.#requestGuestUiString( + { kind: "editor", title: boundPromptTitle("Custom answer: ", question.question) }, + signal, + ); + if (input.kind === "unavailable") return "unavailable"; + // Guest cancelled the Other editor: re-show the select list + // instead of cancelling the whole ask. + if (input.kind === "cancelled") continue; + customInput = input.value; + } else { + selected.add(choice.value); + } + break; } } return { @@ -822,17 +839,14 @@ export class ExtensionUiController { }; } - async #requestGuestUiString( - request: CollabUiRequestDraft, - signal: AbortSignal, - ): Promise { + async #requestGuestUiString(request: CollabUiRequestDraft, signal: AbortSignal): Promise { const host = this.ctx.collabHost; - if (!host) return "unavailable"; + if (!host) return { kind: "unavailable" }; const remote = host.requestGuestUi(request, signal); - if (!remote) return "unavailable"; + if (!remote) return { kind: "unavailable" }; const result = await remote; - if (result.kind === "unavailable") return "unavailable"; - return typeof result.value === "string" ? result.value : undefined; + if (result.kind === "unavailable") return { kind: "unavailable" }; + return typeof result.value === "string" ? { kind: "answered", value: result.value } : { kind: "cancelled" }; } /** diff --git a/packages/coding-agent/test/collab/guest-ui-request.test.ts b/packages/coding-agent/test/collab/guest-ui-request.test.ts index 59806adf3..c1c4a9d8a 100644 --- a/packages/coding-agent/test/collab/guest-ui-request.test.ts +++ b/packages/coding-agent/test/collab/guest-ui-request.test.ts @@ -713,3 +713,45 @@ describe("collab host dialog vs teardown (#4049 follow-up)", () => { } }); }); + +// ── Guest ask "unavailable" literal answer (#4375: tagged guest results) ──── +// +// A guest may legitimately answer with the literal string "unavailable" (e.g. +// a status option). The old `#requestGuestUiString` flattened +// `CollabGuestUiResult` to `string | "unavailable" | undefined`, so that answer +// collided with the transport-unavailable sentinel and cancelled the whole ask +// instead of recording the answer. `CollabHost.requestGuestUi` already returns +// a tagged `CollabGuestUiResult`; this test pins the wire-level contract: a +// guest "unavailable" answer is `{ kind: "answered", value: "unavailable" }`, +// not `{ kind: "unavailable" }`. + +describe("guest ask unavailable literal answer (#4375)", () => { + it("preserves a guest answer of 'unavailable' as answered, not transport-unavailable", async () => { + const ctx = makeHostContext(); + const host = new CollabHost(ctx); + await host.start("ws://localhost:8787"); + ctx.collabHost = host; + try { + const guest = await joinRawGuest(host.link, COLLAB_PROTO); + const welcome = await guest.nextFrame(); + if (welcome.t !== "welcome") throw new Error(`expected welcome, got ${welcome.t}`); + + const pending = host.requestGuestUi({ + kind: "select", + title: "Status?", + options: ["available", "unavailable", "busy"], + }); + if (!pending) throw new Error("expected writable guest UI request"); + const request = await guest.nextFrame(); + if (request.t !== "ui-request") throw new Error(`expected ui-request, got ${request.t}`); + // Guest answers with the literal string "unavailable" — this must be + // treated as a real answer, not a transport-unavailable sentinel. + guest.socket.send({ t: "ui-response", reqId: request.request.reqId, value: "unavailable" }); + const result = await pending; + expect(result).toEqual({ kind: "answered", value: "unavailable" }); + guest.socket.close(); + } finally { + await host.stop("test done"); + } + }); +}); diff --git a/packages/coding-agent/test/modes/components/ask-dialog.test.ts b/packages/coding-agent/test/modes/components/ask-dialog.test.ts index bd2d734c4..39680fccd 100644 --- a/packages/coding-agent/test/modes/components/ask-dialog.test.ts +++ b/packages/coding-agent/test/modes/components/ask-dialog.test.ts @@ -603,10 +603,21 @@ describe("AskDialogComponent", () => { component.handleInput(DOWN); component.handleInput(DOWN); component.handleInput(DOWN); + // Next is disabled when nothing is selected (single-question multi-select), + // so Enter on it must NOT submit an empty result. + component.handleInput(ENTER); + expect(onSubmit).not.toHaveBeenCalled(); + + // Re-select an option to re-enable Next, then submit. + component.handleInput("\x1b[A"); // UP to Other + component.handleInput("\x1b[A"); // UP to Option B + component.handleInput(SPACE); + component.handleInput(DOWN); + component.handleInput(DOWN); component.handleInput(ENTER); expect(onSubmit).toHaveBeenCalledTimes(1); - expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual([]); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option B"]); expect(onSubmit.mock.calls[0][0].results[0].note).toBeUndefined(); }); @@ -992,4 +1003,119 @@ describe("AskDialogComponent", () => { expect(scrolled).not.toContain("Option 01"); expect(scrolled).toContain("Option 29"); }); + + it("single-question multi-select: Next is disabled until an option is selected", () => { + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose multiple?", + options: [{ label: "Option A" }, { label: "Option B" }], + multi: true, + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + + // Row order: [0]=A, [1]=B, [2]=Other, [3]=Next, [4]=Chat + // Navigate to Next (3 DOWNs) and press Enter — must NOT submit empty. + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + expect(onSubmit).not.toHaveBeenCalled(); + + // The Next row should be rendered dimmed (disabled) when nothing is selected. + const output = render(component); + expect(output).toContain("Next"); + + // Select Option A, then Next becomes enabled and submits. + component.handleInput("\x1b[A"); // UP to Other + component.handleInput("\x1b[A"); // UP to Option B + component.handleInput("\x1b[A"); // UP to Option A + component.handleInput(SPACE); + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option A"]); + }); + + it("bounds in-body question header for long multi-line questions", () => { + const onSubmit = vi.fn(); + const longQuestion = "This is a very long question ".repeat(30); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: longQuestion, + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + + // The rendered body must not blow out with the full 30-line question. + // The header is capped to MAX_HEADER_ROWS lines. + const output = render(component); + // The question text should appear but be truncated — verify it does + // not contain the full repeated text (30 copies would be ~870 chars). + expect(output).toContain("This is a very long question"); + // Count occurrences of the repeated phrase — should be far fewer than 30. + const matches = output.match(/This is a very long question/g); + expect(matches?.length ?? 0).toBeLessThan(10); + }); + + it("Other editor cancel returns to the option list without submitting", async () => { + const onPrompt = vi.fn().mockReturnValue(Promise.resolve(undefined)); + const onSubmit = vi.fn(); + const onCancel = vi.fn(); + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Choose one?", + options: [{ label: "Option A" }, { label: "Option B" }], + }, + ]; + + const component = new AskDialogComponent(questions, { + onSubmit, + onCancel, + onChat: vi.fn(), + onPrompt, + }); + + // Navigate to "Other" and press Enter to open the custom input prompt. + component.handleInput(DOWN); + component.handleInput(DOWN); + component.handleInput(ENTER); + await Promise.resolve(); + await Promise.resolve(); + + // The prompt was cancelled (returns undefined). The dialog must stay + // open — no submit, no cancel. + expect(onPrompt).toHaveBeenCalledTimes(1); + expect(onSubmit).not.toHaveBeenCalled(); + expect(onCancel).not.toHaveBeenCalled(); + + // The dialog should still be usable: select Option A and submit. + component.handleInput("\x1b[A"); // UP to Option B + component.handleInput("\x1b[A"); // UP to Option A + component.handleInput(ENTER); + + expect(onSubmit).toHaveBeenCalledTimes(1); + expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option A"]); + }); }); From f66e5276741c1ef907229797e1cf340d0a6a0933 Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Fri, 3 Jul 2026 14:17:31 +0900 Subject: [PATCH 006/205] fix(ask): guest multi-select Next gating, body height bottom border - Omit Next from the guest multi-select ui-request options until at least one option is checked or a custom answer exists, mirroring the local dialog's disabled-Next gating. The remote select has no disabled-row concept, so Next is omitted rather than dimmed (PRRT_kwDOQxs0bc6OFbDW). - Add the bottomBorder(1) term to the ask dialog's fixed-row budget so the rendered dialog no longer overflows the viewport by one row (PRRT_kwDOQxs0bc6OFbDY). - Add focused tests: guest wire-level Next gating round trip, and dialog height <= viewport assertion. --- .../src/modes/components/ask-dialog.ts | 6 +- .../controllers/extension-ui-controller.ts | 15 ++- .../test/collab/guest-ui-request.test.ts | 107 ++++++++++++++++++ .../test/modes/components/ask-dialog.test.ts | 24 ++++ 4 files changed, 149 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/modes/components/ask-dialog.ts b/packages/coding-agent/src/modes/components/ask-dialog.ts index 0b4f5aca0..139843fe1 100644 --- a/packages/coding-agent/src/modes/components/ask-dialog.ts +++ b/packages/coding-agent/src/modes/components/ask-dialog.ts @@ -387,7 +387,11 @@ export class AskDialogComponent implements Component { const height = Math.max(12, process.stdout.rows || 40); const innerWidth = Math.max(1, width - 4); const headerLines = this.#renderHeader(innerWidth); - const fixedRows = 1 + headerLines.length + 1 + 1 + 1; + // topBorder(1) + header(N) + divider(1) + divider(1) + footer(1) + + // bottomBorder(1) = N + 5 fixed rows outside the body. Without the + // bottomBorder term the dialog overflowed the viewport by one row + // (PRRT_kwDOQxs0bc6OFbDY). + const fixedRows = 1 + headerLines.length + 1 + 1 + 1 + 1; const bodyRows = Math.max(MIN_BODY_ROWS, height - fixedRows); const bodyLines = this.#isSubmitTab() ? this.#renderSubmitBody(innerWidth, bodyRows) diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index 9d812e6a2..aa6f170e8 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -760,15 +760,26 @@ export class ExtensionUiController { const checkedIndices = question.options .map((option, index) => (selected.has(option.label) ? index : -1)) .filter(index => index >= 0); + // Mirror the local dialog's Next gating: omit the Next option until + // at least one option is checked or a custom answer exists, so a + // guest cannot submit an empty multi-select result + // (PRRT_kwDOQxs0bc6OFbDW). The remote select has no "disabled" row + // concept, so we omit rather than dim it. + const hasAnswer = selected.size > 0 || customInput !== undefined; + const options = [...baseOptions, ASK_OTHER_OPTION]; + if (hasAnswer) options.push(ASK_NEXT_OPTION); + options.push(ASK_CHAT_OPTION); const choice = await this.#requestGuestUiString( { kind: "select", title: question.question, - options: [...baseOptions, ASK_OTHER_OPTION, ASK_NEXT_OPTION, ASK_CHAT_OPTION], + options, selectionMarker: "checkbox", checkedIndices, markableCount: question.options.length, - helpText: "up/down navigate enter toggle Next → continue esc cancel", + helpText: hasAnswer + ? "up/down navigate enter toggle Next → continue esc cancel" + : "up/down navigate enter toggle esc cancel", }, signal, ); diff --git a/packages/coding-agent/test/collab/guest-ui-request.test.ts b/packages/coding-agent/test/collab/guest-ui-request.test.ts index c1c4a9d8a..76330b0d3 100644 --- a/packages/coding-agent/test/collab/guest-ui-request.test.ts +++ b/packages/coding-agent/test/collab/guest-ui-request.test.ts @@ -24,6 +24,7 @@ import { } from "@oh-my-pi/pi-coding-agent/collab/protocol"; import { CollabSocket } from "@oh-my-pi/pi-coding-agent/collab/relay-client"; import type { + ExtensionAskDialogQuestion, ExtensionUIDialogOptions, ExtensionUISelectItem, } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; @@ -755,3 +756,109 @@ describe("guest ask unavailable literal answer (#4375)", () => { } }); }); + +// ── Guest ask multi-select Next gating (#4375: PRRT_kwDOQxs0bc6OFbDW) ─────── +// +// The local rich dialog disables the Next row on a single-question +// multi-select until at least one option or custom input is chosen. The guest +// mirror has no "disabled row" concept on the wire, so it must OMIT Next from +// the option list until an answer exists, then include it on the next round. +// This test pins that wire-level contract by inspecting consecutive +// ui-request frames. + +/** Context double with the extra members `#showLocalAskDialog` touches when + * mounting the local AskDialogComponent. The local dialog is never driven + * (no input), so it never settles and the remote guest wins the race. + * Reuses makeHostContext for the CollabHost-facing members. */ +function makeAskHostContext(): InteractiveModeContext { + const base = makeHostContext(); + // Stub only the surface the local ask-dialog mount path calls: container + // clear/addChild, ui focus/render, and editor (dispose path). The real + // InteractiveModeContext has many more members; the double-cast below is + // the established test pattern in this file (see makeHostContext) for a + // complex interface that is only partially exercised. + const stub = { + ...base, + editorContainer: { clear: () => {}, addChild: () => {} }, + editor: { getText: () => "", setText: () => {} }, + ui: { + requestRender: () => {}, + setFocus: () => {}, + terminal: { rows: 40, columns: 80 }, + addInputListener: () => () => {}, + }, + }; + return stub as unknown as InteractiveModeContext; +} + +describe("guest ask multi-select Next gating (#4375 PRRT_kwDOQxs0bc6OFbDW)", () => { + /** Skip ui-request-end dismissal frames, wait for the next ui-request. */ + async function nextUiRequest(guest: { + nextFrame(): Promise; + }): Promise { + for (;;) { + const frame = await guest.nextFrame(); + if (frame.t === "ui-request") return frame; + // ui-request-end / other non-request frames are expected between + // rounds; keep draining until the next request arrives. + } + } + + /** Extract string labels from a select ui-request's options, narrowing the + * discriminated union so `options` is visible to the type checker. */ + function selectLabels(frame: CollabFrame & { t: "ui-request" }): string[] { + if (frame.request.kind !== "select") throw new Error(`expected select, got ${frame.request.kind}`); + return frame.request.options.map(o => (typeof o === "string" ? o : o.label)); + } + + it("omits Next from the first ui-request, includes it after a toggle", async () => { + const ctx = makeAskHostContext(); + const host = new CollabHost(ctx); + await host.start("ws://localhost:8787"); + ctx.collabHost = host; + const controller = new ExtensionUiController(ctx); + try { + const guest = await joinRawGuest(host.link, COLLAB_PROTO); + const welcome = await guest.nextFrame(); + if (welcome.t !== "welcome") throw new Error(`expected welcome, got ${welcome.t}`); + + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Pick several?", + options: [{ label: "Option A" }, { label: "Option B" }], + multi: true, + }, + ]; + const result = controller.showAskDialog(questions); + + // First ui-request: Next must be absent (no answer yet). + const first = await nextUiRequest(guest); + const firstLabels = selectLabels(first); + expect(firstLabels).not.toContain("Next →"); + expect(firstLabels).toContain("Option A"); + expect(firstLabels).toContain("Other (type your own)"); + expect(firstLabels).toContain("Chat about this"); + + // Guest toggles Option A — a real answer, not Next/Other/Chat. + guest.socket.send({ t: "ui-response", reqId: first.request.reqId, value: "Option A" }); + + // Second ui-request: Next must now be present. + const second = await nextUiRequest(guest); + const secondLabels = selectLabels(second); + expect(secondLabels).toContain("Next →"); + expect(secondLabels).toContain("Option A"); + + // Guest selects Next to submit. + guest.socket.send({ t: "ui-response", reqId: second.request.reqId, value: "Next →" }); + const settled = await result; + expect(settled?.kind).toBe("submit"); + if (settled?.kind === "submit") { + expect(settled.results[0]?.selectedOptions).toEqual(["Option A"]); + } + guest.socket.close(); + } finally { + await host.stop("test done"); + } + }); +}); diff --git a/packages/coding-agent/test/modes/components/ask-dialog.test.ts b/packages/coding-agent/test/modes/components/ask-dialog.test.ts index 39680fccd..b90c7f406 100644 --- a/packages/coding-agent/test/modes/components/ask-dialog.test.ts +++ b/packages/coding-agent/test/modes/components/ask-dialog.test.ts @@ -1118,4 +1118,28 @@ describe("AskDialogComponent", () => { expect(onSubmit).toHaveBeenCalledTimes(1); expect(onSubmit.mock.calls[0][0].results[0].selectedOptions).toEqual(["Option A"]); }); + + it("rendered dialog height fits within the viewport (bottom border included)", () => { + // PRRT_kwDOQxs0bc6OFbDY: the fixed-row budget must account for the + // bottom border too. Before the fix, fixedRows omitted bottomBorder(1), + // so the dialog rendered height+1 lines — overflowing the viewport by + // one row. After the fix, total lines <= height. + const questions: ExtensionAskDialogQuestion[] = [ + { + id: "q1", + question: "Pick one?", + options: [{ label: "Option A" }, { label: "Option B" }], + multi: true, + }, + ]; + const component = new AskDialogComponent(questions, { + onSubmit: vi.fn(), + onCancel: vi.fn(), + onChat: vi.fn(), + onPrompt: vi.fn(), + }); + const height = Math.max(12, process.stdout.rows || 40); + const lines = component.render(80); + expect(lines.length).toBeLessThanOrEqual(height); + }); }); From 71c9cc0324455ec7ca9834adbba3a797455854a7 Mon Sep 17 00:00:00 2001 From: ben Date: Thu, 9 Jul 2026 16:02:23 +0800 Subject: [PATCH 007/205] fix(coding-agent): keep same-realm setCwd from killing TUI sessions When the JS eval worker falls back to the in-process inline path, concurrent JsRuntime instances share one realm. setCwd used to throw on exclusive-owner conflicts, and the microtask delivery path turned that into a fatal unhandledRejection that postmortem exited on. Stamp local cwd without stealing the active realm, report init failures over the worker protocol, and cover process survival with in-process and child-process regressions. --- packages/coding-agent/CHANGELOG.md | 4 + .../src/eval/js/shared/runtime.ts | 16 +- .../coding-agent/src/eval/js/worker-core.ts | 11 +- .../src/tools/browser/cmux/cmux-tab.ts | 28 +- .../test/eval/runtime-global-dispose.test.ts | 31 ++- .../test/eval/worker-core.test.ts | 256 +++++++++++++++++- 6 files changed, 325 insertions(+), 21 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f9b8bb0ef..6be6a947b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed interactive TUI sessions dying with `Unhandled rejection: Cannot set cwd while another same-realm JS runtime is running` after the JS eval worker fell back to the in-process inline path (commonly when the worker could not load `pi_natives`). Concurrent inline eval/browser runtimes now stamp cwd without stealing the exclusive realm; exclusive activation remains on `run`/`setRunScope`, and WorkerCore `init` reports failures via `init-failed` instead of throwing out of the microtask path. + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/coding-agent/src/eval/js/shared/runtime.ts b/packages/coding-agent/src/eval/js/shared/runtime.ts index 1d88465be..6af513780 100644 --- a/packages/coding-agent/src/eval/js/shared/runtime.ts +++ b/packages/coding-agent/src/eval/js/shared/runtime.ts @@ -189,8 +189,22 @@ export class JsRuntime { } setCwd(cwd: string): void { - this.#activateGlobals("set cwd"); + // Always stamp the local field: WorkerCore/browser/cmux call setCwd from + // init and pre-run paths that may race another same-realm runtime. The + // exclusive global bag is only needed when this runtime is about to + // execute; run()/setRunScope still assert ownership. A throw here used + // to escape via the inline-worker microtask path as a fatal + // unhandledRejection and kill the whole interactive session. + if (this.#disposed) throw new Error("Cannot set cwd on a disposed JS runtime"); this.#cwd = cwd; + try { + this.#activateGlobals("set cwd"); + } catch (err) { + if (err instanceof Error && err.message.includes("another same-realm JS runtime is running")) { + return; + } + throw err; + } const session = (globalThis as { __omp_session__?: { cwd?: string } }).__omp_session__; if (session) session.cwd = cwd; } diff --git a/packages/coding-agent/src/eval/js/worker-core.ts b/packages/coding-agent/src/eval/js/worker-core.ts index c23517ee7..f288e5a6a 100644 --- a/packages/coding-agent/src/eval/js/worker-core.ts +++ b/packages/coding-agent/src/eval/js/worker-core.ts @@ -184,8 +184,15 @@ export class WorkerCore { #handle(msg: WorkerInbound): void { switch (msg.type) { case "init": - this.#ensureRuntime(msg.snapshot); - this.#transport.send({ type: "ready" }); + try { + this.#ensureRuntime(msg.snapshot); + this.#transport.send({ type: "ready" }); + } catch (error) { + // Inline fallback delivers messages on a microtask. A sync throw + // from ensureRuntime/setCwd would otherwise become a process-fatal + // unhandledRejection on the main thread. + this.#transport.send({ type: "init-failed", error: errorPayload(error) }); + } return; case "run": void this.#runOne(msg.runId, msg.code, msg.filename, msg.snapshot); diff --git a/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts b/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts index 0616812fc..5a25bac66 100644 --- a/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts +++ b/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts @@ -1288,18 +1288,6 @@ export async function runCmuxCode(tab: CmuxTab, opts: RunCmuxCodeOptions): Promi const screenshots: ScreenshotResult[] = []; const runId = crypto.randomUUID(); tab.setRunContext({ session: opts.snapshot, displays, screenshots, signal, timeoutMs: opts.timeoutMs }); - const runtime = tab.ensureRuntime(opts.snapshot); - runtime.setCwd(opts.snapshot.cwd); - const runTab = bindBrowserRunFacade(tab, signal); - runtime.setRunScope({ - page: bindBrowserRunFacade(tab.page, signal), - browser: bindBrowserRunFacade(tab.browser, signal), - tab: runTab, - assert: (cond: unknown, text?: string): void => { - if (!cond) throw new ToolError(text ?? "Assertion failed"); - }, - wait: (ms: number): Promise => waitForBrowserRun(ms, signal), - }); const { promise: cancelRejection, reject } = Promise.withResolvers(); const onAbort = (): void => { @@ -1317,6 +1305,22 @@ export async function runCmuxCode(tab: CmuxTab, opts: RunCmuxCodeOptions): Promi else signal.addEventListener("abort", onAbort, { once: true }); try { + const runtime = tab.ensureRuntime(opts.snapshot); + // setCwd is non-exclusive; setRunScope/run still assert same-realm ownership. + // Keep both inside try so a concurrent in-process eval/browser run surfaces as + // a rejected promise the supervisor can report, never an unhandled rejection. + runtime.setCwd(opts.snapshot.cwd); + const runTab = bindBrowserRunFacade(tab, signal); + runtime.setRunScope({ + page: bindBrowserRunFacade(tab.page, signal), + browser: bindBrowserRunFacade(tab.browser, signal), + tab: runTab, + assert: (cond: unknown, text?: string): void => { + if (!cond) throw new ToolError(text ?? "Assertion failed"); + }, + wait: (ms: number): Promise => waitForBrowserRun(ms, signal), + }); + const hooks: RuntimeHooks = { onText: chunk => { throwIfAborted(signal); diff --git a/packages/coding-agent/test/eval/runtime-global-dispose.test.ts b/packages/coding-agent/test/eval/runtime-global-dispose.test.ts index 0ec659c92..5bf906630 100644 --- a/packages/coding-agent/test/eval/runtime-global-dispose.test.ts +++ b/packages/coding-agent/test/eval/runtime-global-dispose.test.ts @@ -96,18 +96,24 @@ describe("JsRuntime global disposal", () => { } }); - it("rejects cross-runtime mutations while another same-realm runtime is running", async () => { + it("defers cross-runtime setCwd while another same-realm runtime is running", async () => { const before = snapshotGlobals(); const globals = globalThis as Record; - const first = new JsRuntime({ initialCwd: process.cwd(), sessionId: "first-overlap" }); - const second = new JsRuntime({ initialCwd: process.cwd(), sessionId: "second-overlap" }); + const firstCwd = process.cwd(); + const secondCwd = process.cwd(); + const first = new JsRuntime({ initialCwd: firstCwd, sessionId: "first-overlap" }); + const second = new JsRuntime({ initialCwd: secondCwd, sessionId: "second-overlap" }); const gate = Promise.withResolvers(); let activeSecond: Promise | undefined; + const pendingCwd = `${firstCwd}/pending-same-realm-cwd`; try { second.setRunScope({ gate: gate.promise }); activeSecond = second.run("await gate;", undefined, hooks); - expect(() => first.setCwd(process.cwd())).toThrow("another same-realm JS runtime is running"); + // Local cwd may be stamped without stealing the active realm. + first.setCwd(pendingCwd); + expect(first.cwd).toBe(pendingCwd); + expect(globals.__omp_helpers__).toBe(second.helpers); await first.run("1", undefined, hooks).then( () => { throw new Error("expected active runtime rejection"); @@ -120,8 +126,12 @@ describe("JsRuntime global disposal", () => { ); gate.resolve(); await activeSecond; - first.setCwd(process.cwd()); + // After the exclusive run ends, setCwd can promote globals and keep the stamped cwd. + first.setCwd(pendingCwd); + expect(first.cwd).toBe(pendingCwd); expect(globals.__omp_helpers__).toBe(first.helpers); + const session = globals.__omp_session__ as { cwd?: string } | undefined; + expect(session?.cwd).toBe(pendingCwd); } finally { gate.resolve(); if (activeSecond) await activeSecond.catch(() => undefined); @@ -131,4 +141,15 @@ describe("JsRuntime global disposal", () => { restoreGlobals(before); } }); + + it("setCwd on a disposed runtime still throws", () => { + const before = snapshotGlobals(); + const runtime = new JsRuntime({ initialCwd: process.cwd(), sessionId: "disposed-setcwd" }); + try { + runtime.dispose(); + expect(() => runtime.setCwd(process.cwd())).toThrow("Cannot set cwd on a disposed JS runtime"); + } finally { + restoreGlobals(before); + } + }); }); diff --git a/packages/coding-agent/test/eval/worker-core.test.ts b/packages/coding-agent/test/eval/worker-core.test.ts index ab1d71846..42cac4cb8 100644 --- a/packages/coding-agent/test/eval/worker-core.test.ts +++ b/packages/coding-agent/test/eval/worker-core.test.ts @@ -1,4 +1,8 @@ import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { pathToFileURL } from "node:url"; import { WorkerCore } from "@oh-my-pi/pi-coding-agent/eval/js/worker-core"; import type { SessionSnapshot, @@ -61,6 +65,28 @@ async function initializeWorker(harness: WorkerHarness, snapshot: SessionSnapsho expect((await ready).type).toBe("ready"); } +function installFatalCapture(): { + fatal: unknown[]; + uninstall: () => void; +} { + const fatal: unknown[] = []; + const onUnhandled = (reason: unknown): void => { + fatal.push(reason); + }; + const onUncaught = (err: Error): void => { + fatal.push(err); + }; + process.on("unhandledRejection", onUnhandled); + process.on("uncaughtException", onUncaught); + return { + fatal, + uninstall: () => { + process.off("unhandledRejection", onUnhandled); + process.off("uncaughtException", onUncaught); + }, + }; +} + describe("WorkerCore", () => { it("reports same-realm cwd conflicts through the worker protocol", async () => { const first = createWorkerHarness(); @@ -101,7 +127,7 @@ describe("WorkerCore", () => { type: "result", runId: "overlap-second-runtime", ok: false, - error: { message: "Cannot set cwd while another same-realm JS runtime is running" }, + error: { message: "Cannot run code while another same-realm JS runtime is running" }, }); } finally { gate.resolve(); @@ -111,4 +137,232 @@ describe("WorkerCore", () => { second.send({ type: "close" }); } }); + + it("re-init while a same-realm run is live does not crash the process", async () => { + const first = createWorkerHarness(); + const second = createWorkerHarness(); + const cwd = process.cwd(); + await initializeWorker(first, { cwd, sessionId: "reinit-first", localRoots: {} }); + await initializeWorker(second, { cwd, sessionId: "reinit-second", localRoots: {} }); + + const gate = Promise.withResolvers(); + const entered = Promise.withResolvers(); + (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }).__omp_worker_core_gate = { + entered: () => entered.resolve(), + wait: gate.promise, + }; + + const { fatal, uninstall } = installFatalCapture(); + try { + first.send({ + type: "run", + runId: "hold-for-reinit", + code: "globalThis.__omp_worker_core_gate.entered(); await globalThis.__omp_worker_core_gate.wait;", + filename: "[reinit-first].js", + snapshot: { cwd, sessionId: "reinit-first", localRoots: {} }, + }); + await entered.promise; + + // Re-init the second core while the first still owns the realm. Production + // inline workers deliver this on a microtask; a setCwd throw here used to + // become a process-fatal unhandledRejection / uncaughtException. + const reinit = waitForMessage(second, message => message.type === "ready" || message.type === "init-failed"); + second.send({ type: "init", snapshot: { cwd, sessionId: "reinit-second", localRoots: {} } }); + const reply = await reinit; + expect(reply.type).toBe("ready"); + + // Concurrent run still fails at the exclusive run boundary, via protocol. + const result = waitForMessage( + second, + message => message.type === "result" && message.runId === "overlap-after-reinit", + ); + second.send({ + type: "run", + runId: "overlap-after-reinit", + code: "1 + 1;", + filename: "[reinit-second].js", + snapshot: { cwd, sessionId: "reinit-second", localRoots: {} }, + }); + expect(await result).toMatchObject({ + type: "result", + runId: "overlap-after-reinit", + ok: false, + error: { message: "Cannot run code while another same-realm JS runtime is running" }, + }); + + // Drain microtasks so a latent fatal would surface. + await Bun.sleep(0); + expect(fatal).toEqual([]); + } finally { + uninstall(); + gate.resolve(); + delete (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }) + .__omp_worker_core_gate; + first.send({ type: "close" }); + second.send({ type: "close" }); + } + }); + + it("concurrent inits under a live same-realm run stay process-safe", async () => { + const first = createWorkerHarness(); + const second = createWorkerHarness(); + const third = createWorkerHarness(); + const cwd = process.cwd(); + await initializeWorker(first, { cwd, sessionId: "init-live-first", localRoots: {} }); + await initializeWorker(second, { cwd, sessionId: "init-live-second", localRoots: {} }); + await initializeWorker(third, { cwd, sessionId: "init-live-third", localRoots: {} }); + + const gate = Promise.withResolvers(); + const entered = Promise.withResolvers(); + (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }).__omp_worker_core_gate = { + entered: () => entered.resolve(), + wait: gate.promise, + }; + + const { fatal, uninstall } = installFatalCapture(); + try { + first.send({ + type: "run", + runId: "hold-for-multi-init", + code: "globalThis.__omp_worker_core_gate.entered(); await globalThis.__omp_worker_core_gate.wait;", + filename: "[init-live-first].js", + snapshot: { cwd, sessionId: "init-live-first", localRoots: {} }, + }); + await entered.promise; + + const readySecond = waitForMessage( + second, + message => message.type === "ready" || message.type === "init-failed", + ); + const readyThird = waitForMessage( + third, + message => message.type === "ready" || message.type === "init-failed", + ); + second.send({ type: "init", snapshot: { cwd, sessionId: "init-live-second", localRoots: {} } }); + third.send({ type: "init", snapshot: { cwd, sessionId: "init-live-third", localRoots: {} } }); + expect((await readySecond).type).toBe("ready"); + expect((await readyThird).type).toBe("ready"); + + const resultSecond = waitForMessage( + second, + message => message.type === "result" && message.runId === "overlap-second", + ); + const resultThird = waitForMessage( + third, + message => message.type === "result" && message.runId === "overlap-third", + ); + second.send({ + type: "run", + runId: "overlap-second", + code: "2", + filename: "[init-live-second].js", + snapshot: { cwd, sessionId: "init-live-second", localRoots: {} }, + }); + third.send({ + type: "run", + runId: "overlap-third", + code: "3", + filename: "[init-live-third].js", + snapshot: { cwd, sessionId: "init-live-third", localRoots: {} }, + }); + expect(await resultSecond).toMatchObject({ + type: "result", + runId: "overlap-second", + ok: false, + error: { message: "Cannot run code while another same-realm JS runtime is running" }, + }); + expect(await resultThird).toMatchObject({ + type: "result", + runId: "overlap-third", + ok: false, + error: { message: "Cannot run code while another same-realm JS runtime is running" }, + }); + + await Bun.sleep(0); + expect(fatal).toEqual([]); + } finally { + uninstall(); + gate.resolve(); + delete (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }) + .__omp_worker_core_gate; + first.send({ type: "close" }); + second.send({ type: "close" }); + third.send({ type: "close" }); + } + }); + + it("survives concurrent same-realm setCwd in a child process with postmortem loaded", async () => { + // Process-level oracle: the production crash was postmortem killing the process + // after an unhandled rejection from concurrent inline setCwd. This must stay green + // even when postmortem's fatal handlers are installed. + const postmortemUrl = pathToFileURL(path.resolve(import.meta.dir, "../../../utils/src/postmortem.ts")).href; + const runtimeUrl = pathToFileURL(path.resolve(import.meta.dir, "../../src/eval/js/shared/runtime.ts")).href; + + const probe = `import { pathToFileURL } from "node:url"; + +await import(${JSON.stringify(postmortemUrl)}); +const { JsRuntime } = await import(${JSON.stringify(runtimeUrl)}); + +const first = new JsRuntime({ initialCwd: process.cwd(), sessionId: "child-first" }); +const second = new JsRuntime({ initialCwd: process.cwd(), sessionId: "child-second" }); +const gate = Promise.withResolvers(); +const entered = Promise.withResolvers(); + +const hooks = { + onText() {}, + onDisplay() {}, + callTool: async () => undefined, +}; + +second.setRunScope({ gate: gate.promise, entered: () => entered.resolve() }); +const hold = second.run("entered(); await gate;", "[child-second].js", hooks); +await entered.promise; + +// Historical crash path: concurrent setCwd while another same-realm runtime is live. +first.setCwd(process.cwd() + "/child-pending"); +second.setCwd(process.cwd()); + +// Microtask delivery must not become process-fatal either. +queueMicrotask(() => { + first.setCwd(process.cwd() + "/child-pending-2"); +}); +await Promise.resolve(); +await Bun.sleep(0); + +gate.resolve(); +await hold; +first.dispose(); +second.dispose(); +console.log("survived concurrent setCwd"); +process.exit(0); +`; + + const root = await fs.mkdtemp(path.join(os.tmpdir(), "omp-same-realm-")); + const probePath = path.join(root, "probe.ts"); + try { + await Bun.write(probePath, probe); + const proc = Bun.spawn([process.execPath, probePath], { + cwd: process.cwd(), + stdout: "pipe", + stderr: "pipe", + env: { ...process.env }, + }); + const watchdog = Bun.sleep(5000).then(() => { + proc.kill(); + return -999; + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + Promise.race([proc.exited, watchdog]), + ]); + expect(exitCode).toBe(0); + expect(stdout).toContain("survived concurrent setCwd"); + expect(stderr).not.toContain("[Unhandled Rejection]"); + expect(stderr).not.toContain("[Uncaught Exception]"); + expect(stderr).not.toContain("another same-realm JS runtime is running"); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + }); }); From d8d33d4e8c4185569ba07bd31ae7244689362733 Mon Sep 17 00:00:00 2001 From: ben Date: Thu, 9 Jul 2026 18:02:44 +0800 Subject: [PATCH 008/205] fix(ai): rotate xai-oauth on SuperGrok credit exhaustion xAI returns HTTP 403 with "run out of credits" / spending-limit when an account is parked. Classify that as a usage limit so multi-account pools switch siblings instead of sticking to the exhausted credential. --- packages/ai/CHANGELOG.md | 4 +++ packages/ai/src/error/rate-limit.ts | 8 +++++- packages/ai/test/auth-retry.test.ts | 12 +++++++++ .../auth-storage-force-refresh-rotate.test.ts | 26 +++++++++++++++++++ packages/ai/test/rate-limit-utils.test.ts | 19 ++++++++++++++ 5 files changed, 68 insertions(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3b40e8bea..d30b103e4 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed xAI SuperGrok multi-account rotation when an account returns HTTP 403 `run out of credits` / `personal-team-blocked:spending-limit`. That account-local cap is now classified as a usage limit so `streamSimple` auth-retry and `rotateSessionCredential` switch to a sibling `xai-oauth` credential instead of sticking to the exhausted account. + ### Changed - Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). diff --git a/packages/ai/src/error/rate-limit.ts b/packages/ai/src/error/rate-limit.ts index fb5678188..cef4dca02 100644 --- a/packages/ai/src/error/rate-limit.ts +++ b/packages/ai/src/error/rate-limit.ts @@ -67,6 +67,12 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason { lower.includes("exhausted") || lower.includes("quota") || lower.includes("usage limit") || + // xAI SuperGrok: HTTP 403 "run out of credits" / spending-limit is an + // account-local cap — rotate, don't treat as auth failure. + lower.includes("run out of credits") || + lower.includes("out of credits") || + lower.includes("spending-limit") || + lower.includes("spending limit") || INSUFFICIENT_BALANCE_PATTERN.test(errorMessage) ) { return "QUOTA_EXHAUSTED"; @@ -100,7 +106,7 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number { /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */ const USAGE_LIMIT_PATTERN = - /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?(?:exceeded|reached|insufficient)|额度不足|额度耗尽|resource.?exhausted|exhausted your capacity|quota will reset|insufficient.?(?:balance|quota)/i; + /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?(?:exceeded|reached|insufficient)|额度不足|额度耗尽|resource.?exhausted|exhausted your capacity|quota will reset|insufficient.?(?:balance|quota)|run out of credits|out of credits|spending[- _]?limit|personal-team-blocked/i; /** * HTTP status codes that, absent richer body classification, represent an diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts index 44a4a9f89..c851d8f3b 100644 --- a/packages/ai/test/auth-retry.test.ts +++ b/packages/ai/test/auth-retry.test.ts @@ -52,6 +52,18 @@ describe("isAuthRetryableError", () => { // credentials won't help an org/global limit. expect(isAuthRetryableError(Object.assign(new Error("429 too many requests"), { status: 429 }))).toBe(false); expect(isAuthRetryableError("Error: 401 unauthorized")).toBe(true); + // xAI SuperGrok surfaces account exhaustion as 403 + "run out of credits" / + // spending-limit, not 429. Must rotate so multi-account xai-oauth pools work. + expect( + isAuthRetryableError( + Object.assign( + new Error( + "403 You have run out of credits or need a Grok subscription. Add credits at https://grok.com/?_s=usage or upgrade at https://grok.com/supergrok. (type=personal-team-blocked:spending-limit)", + ), + { status: 403 }, + ), + ), + ).toBe(true); expect(isAuthRetryableError(authError(403))).toBe(false); expect(isAuthRetryableError(authError(500))).toBe(false); expect(isAuthRetryableError(new Error("network blip"))).toBe(false); diff --git a/packages/ai/test/auth-storage-force-refresh-rotate.test.ts b/packages/ai/test/auth-storage-force-refresh-rotate.test.ts index 1d01f76e4..94856c83e 100644 --- a/packages/ai/test/auth-storage-force-refresh-rotate.test.ts +++ b/packages/ai/test/auth-storage-force-refresh-rotate.test.ts @@ -150,6 +150,32 @@ describe("AuthStorage forceRefresh + rotateSessionCredential", () => { expect(second).not.toBe(first); }); + test("rotateSessionCredential(xAI credits 403) blocks the exhausted account and rotates", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }, + { type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() }, + ]); + + const first = await authStorage.getApiKey(PROVIDER, "xai-credits"); + const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached"); + const xaiCreditsError = Object.assign( + new Error( + "403 You have run out of credits or need a Grok subscription. Add credits at https://grok.com/?_s=usage or upgrade at https://grok.com/supergrok. (type=personal-team-blocked:spending-limit)", + ), + { status: 403 }, + ); + + const rotated = await authStorage.rotateSessionCredential(PROVIDER, "xai-credits", { + error: xaiCreditsError, + }); + + expect(rotated).toBe(true); + expect(usageLimitSpy).toHaveBeenCalledTimes(1); + expect(await authStorage.getApiKey(PROVIDER, "xai-credits")).not.toBe(first); + }); + test("rotateSessionCredential treats quota payloads as temporary usage blocks", async () => { if (!authStorage) throw new Error("test setup failed"); registerProvider(); diff --git a/packages/ai/test/rate-limit-utils.test.ts b/packages/ai/test/rate-limit-utils.test.ts index ef0713a45..b2c9f57dd 100644 --- a/packages/ai/test/rate-limit-utils.test.ts +++ b/packages/ai/test/rate-limit-utils.test.ts @@ -131,6 +131,17 @@ describe("isUsageLimit", () => { expect(isUsageLimit("额度耗尽")).toBe(true); }); + it("detects xAI Grok SuperGrok credit exhaustion as a credential-rotatable usage limit", () => { + // xAI returns HTTP 403 with (type=personal-team-blocked:spending-limit), not a + // 429 usage_limit_reached. Without this match, multi-account xai-oauth pools + // stick to the exhausted credential instead of rotating siblings. + const message = + "403 You have run out of credits or need a Grok subscription. Add credits at https://grok.com/?_s=usage or upgrade at https://grok.com/supergrok.\nYou have run out of credits or need a Grok subscription. Add credits at https://grok.com/?_s=usage or upgrade at https://grok.com/supergrok. (type=personal-team-blocked:spending-limit)"; + expect(isUsageLimit(message)).toBe(true); + expect(isUsageLimit(Object.assign(new Error(message), { status: 403 }))).toBe(true); + expect(parseRateLimitReason(message)).toBe("QUOTA_EXHAUSTED"); + }); + it("detects OpenAI quota payload codes as credential-rotatable usage limits", () => { for (const message of ["insufficient_quota", "usage_limit_exceeded", "usage_limit_reached"]) { expect(isUsageLimit(message)).toBe(true); @@ -184,6 +195,14 @@ describe("isUsageLimitOutcome", () => { ).toBe(true); }); + it("rotates on xAI Grok 403 credit/spending-limit exhaustion regardless of status", () => { + const message = + "403 You have run out of credits or need a Grok subscription. Add credits at https://grok.com/?_s=usage or upgrade at https://grok.com/supergrok. (type=personal-team-blocked:spending-limit)"; + expect(isUsageLimitOutcome(403, message)).toBe(true); + expect(isUsageLimitOutcome(undefined, message)).toBe(true); + expect(isUsageLimitOutcome(429, message)).toBe(true); + }); + it("does not rotate on auth/invalid-request statuses with unrelated bodies", () => { expect(isUsageLimitOutcome(401, "Invalid API key")).toBe(false); expect(isUsageLimitOutcome(400, "invalid_request_error: model unsupported")).toBe(false); From 0f3cf73a854ced718fd2fac27e5337c2bc3c8e9d Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 19:12:48 +0000 Subject: [PATCH 009/205] fix(agent): stopped terminal yield wake loops - Aborted the active agent loop synchronously when a terminal yield tool result finishes, so IRC-wake turns stop before another provider call. - Added a regression covering idle IRC wake handling after a terminal yield. Fixes #4963 --- packages/agent/CHANGELOG.md | 4 + packages/agent/src/agent-loop.ts | 6 ++ packages/coding-agent/CHANGELOG.md | 4 + .../src/prompts/system/workflow-notice.md | 10 +- .../coding-agent/src/session/agent-session.ts | 33 +++++- .../agent-session-advisor-suppression.test.ts | 100 +++++++++++++++++- 6 files changed, 148 insertions(+), 9 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 04b655eeb..a93ed021e 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed aborted tool-result hooks from continuing into another provider call before the abort settled. ([#4963](https://github.com/can1357/oh-my-pi/issues/4963)) + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index c8b1b82e5..9d7f33230 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -1084,6 +1084,12 @@ async function runLoopBody( } } + // A tool hook may abort to mark the tool result as terminal (e.g. subagent yield). + // Stop before the next provider call; event listeners observe the result too late. + if (signal?.aborted) { + hasMoreToolCalls = false; + } + if (toolCalls.length > 0) { pausedTurnContinuations = 0; } else if ( diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d1ad895e4..bac444188 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed kept-alive task subagents entering a repeated provider-call loop after an IRC wake and terminal `yield`. ([#4963](https://github.com/can1357/oh-my-pi/issues/4963)) + ## [16.3.13] - 2026-07-09 ### Fixed diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 74eddee21..3e62ca5b0 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -51,20 +51,20 @@ Decompose first, then {{#if taskBatch}}batch the independent leaves{{else}}issue {{#if taskBatch}} task( - context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...", + context: "# Goal\nReview the auth diff…\n# Constraints\nRead-only…\n# Contract\nReturn findings as severity/file/line/fix…", tasks: [ - { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection...\n# Acceptance\nReturn confirmed findings only..." }, - { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." }, + { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection…\n# Acceptance\nReturn confirmed findings only…" }, + { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance…\n# Acceptance\nReturn mismatches and exact prompt lines…" }, ] ) {{else}} task( role: "Auth Storage Reviewer", - assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only..." + assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only…" ) task( role: "Prompt Contract Reviewer", - assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only..." + assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only…" ) {{/if}} diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..9fbe9f3a0 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2222,8 +2222,8 @@ export class AgentSession { this.#maybeAbortStreamingEdit(event); this.#maybeInterruptGeminiHeaderRunaway(message, assistantMessageEvent); }); - // Per-tool TTSR reminders are folded into the matched tool's result via this hook. - this.agent.afterToolCall = ctx => this.#ttsrAfterToolCall(ctx); + // Tool-result hook owns synchronous post-tool actions that must affect the current loop. + this.agent.afterToolCall = ctx => this.#afterToolCall(ctx); this.agent.providerSessionState = this.#providerSessionState; this.#syncAgentSessionId(); this.#syncTodoPhasesFromBranch(); @@ -3671,9 +3671,10 @@ export class AgentSession { this.#planModeReminderAwaitingProgress = false; } } - if (event.type === "tool_execution_end" && event.toolName === "yield" && !event.isError) { + if (event.type === "tool_execution_end" && this.#isTerminalYieldToolResult(event)) { this.#lastSuccessfulYieldToolCallId = event.toolCallId; this.#yieldTerminationPending = true; + this.agent.abort(); } // TTSR: Check for pattern matches on assistant text/thinking and tool argument deltas @@ -4418,6 +4419,19 @@ export class AgentSession { } } + #afterToolCall(ctx: AfterToolCallContext): AfterToolCallResult | undefined { + if ( + this.#isTerminalYieldToolResult({ + toolName: ctx.toolCall.name, + isError: ctx.isError, + result: ctx.result, + }) + ) { + this.agent.abort(); + } + return this.#ttsrAfterToolCall(ctx); + } + /** `afterToolCall` hook: fold any per-tool TTSR reminders into the result. */ #ttsrAfterToolCall(ctx: AfterToolCallContext): AfterToolCallResult | undefined { const rules = this.#perToolTtsrInjections.get(ctx.toolCall.id); @@ -10589,6 +10603,19 @@ export class AgentSession { } return COMPACTION_CHECK_NONE; } + #isTerminalYieldToolResult(event: { toolName: string; isError?: boolean; result?: { details?: unknown } }): boolean { + if (event.toolName !== "yield" || event.isError) return false; + const details = event.result?.details; + if (!details || typeof details !== "object") return true; + const record = details as Record; + return !( + record.status === "success" && + Array.isArray(record.type) && + record.type.length > 0 && + record.type.every(item => typeof item === "string") + ); + } + #assistantMessageHasSuccessfulYieldToolCall(assistantMessage: AssistantMessage, toolCallId: string): boolean { const lastToolCall = assistantMessage.content .slice() diff --git a/packages/coding-agent/test/agent-session-advisor-suppression.test.ts b/packages/coding-agent/test/agent-session-advisor-suppression.test.ts index 2ac1780ed..363232740 100644 --- a/packages/coding-agent/test/agent-session-advisor-suppression.test.ts +++ b/packages/coding-agent/test/agent-session-advisor-suppression.test.ts @@ -18,7 +18,8 @@ * follow-up stays queued for the next explicit resume rather than auto-running. */ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { Agent, type AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { Agent, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import type { ToolCall } from "@oh-my-pi/pi-ai"; import { createMockModel, type MockModel, type MockResponse } from "@oh-my-pi/pi-ai/providers/mock"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -29,6 +30,18 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { USER_INTERRUPT_LABEL } from "@oh-my-pi/pi-coding-agent/session/messages"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { Snowflake, TempDir } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; + +interface MockYieldDetails { + status: "success"; + data?: unknown; + type?: string | string[]; +} + +const mockYieldParameters = type({ + result: "unknown", + "type?": "unknown", +}); const ADVISOR_TYPE = "advisor"; @@ -94,6 +107,56 @@ describe("AgentSession advisor auto-resume suppression", () => { return { session, sessionManager, mock, streamStarted: started.promise }; } + function readYieldResultData(result: unknown): unknown { + if (!result || typeof result !== "object" || !("data" in result)) return undefined; + return result.data; + } + + function isYieldType(value: unknown): value is string | string[] { + return ( + typeof value === "string" || + (Array.isArray(value) && value.length > 0 && value.every(item => typeof item === "string")) + ); + } + + function createMockYieldTool(): AgentTool { + return { + name: "yield", + label: "Yield", + description: "Mock yield tool", + parameters: mockYieldParameters, + execute: async (_toolCallId, params) => { + const details: MockYieldDetails = { status: "success", data: readYieldResultData(params.result) }; + if (isYieldType(params.type)) details.type = params.type; + return { + content: [{ type: "text", text: "Result submitted." }], + details, + }; + }, + }; + } + + function createYieldMockResponse(args: { result: { data: unknown }; type?: string | string[] }): MockResponse { + const toolCall: ToolCall = { + type: "toolCall", + id: `call_yield_${Snowflake.next()}`, + name: "yield", + arguments: args, + }; + return { + content: [toolCall], + stopReason: "toolUse", + usage: { + input: 1, + output: 1, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 2, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + }; + } + function advisorCard(content: string) { return { customType: ADVISOR_TYPE, @@ -325,6 +388,41 @@ describe("AgentSession advisor auto-resume suppression", () => { expect(mock.calls.length).toBe(2); }); + it("stops an idle IRC wake after a terminal yield", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) throw new Error("Expected claude-sonnet-4-5 model to exist"); + let providerCalls = 0; + const mock = createMockModel({ + handler: () => { + providerCalls++; + if (providerCalls > 1) { + throw new Error("terminal yield must not start a second provider call"); + } + return createYieldMockResponse({ result: { data: { ok: true } } }); + }, + }); + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [createMockYieldTool()] }, + streamFn: mock.stream, + }); + const sessionManager = SessionManager.inMemory(); + const settings = Settings.isolated({ "compaction.enabled": false }); + const authStorage = await AuthStorage.create(tempDir.join(`auth-${Snowflake.next()}.db`)); + authStorages.push(authStorage); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, tempDir.join("models.yml")); + session = new AgentSession({ agent, sessionManager, settings, modelRegistry }); + const msg: IrcMessage = { id: "m-yield", from: "peer", to: "me", body: "status?", ts: Date.now() }; + + const outcome = await session.deliverIrcMessage(msg); + await session.waitForIdle(); + + expect(outcome).toBe("woken"); + expect(providerCalls).toBe(1); + expect(mock.calls.length).toBe(1); + }); + it("flushes an accepted IRC aside on dispose instead of dropping it", async () => { const { session, streamStarted } = await createParkedSession(); const running = session.prompt("do the thing"); From 92307e11d87b55787694190685d3678307408969 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 19:24:39 +0000 Subject: [PATCH 010/205] test(agent): updated terminal yield expectations - Updated stale yield-empty-stop regressions for the new terminal-yield contract. - Kept coverage for clearing yield termination before the next prompt and IRC wake. Fixes #4963 --- ...ssion-yield-empty-stop-suppression.test.ts | 40 ++++++++----------- 1 file changed, 16 insertions(+), 24 deletions(-) diff --git a/packages/coding-agent/test/agent-session-yield-empty-stop-suppression.test.ts b/packages/coding-agent/test/agent-session-yield-empty-stop-suppression.test.ts index f414402b4..7d8abc77d 100644 --- a/packages/coding-agent/test/agent-session-yield-empty-stop-suppression.test.ts +++ b/packages/coding-agent/test/agent-session-yield-empty-stop-suppression.test.ts @@ -1,11 +1,11 @@ /** - * Regression: a trailing empty assistant `stop` arriving after a successful - * `yield` must NOT trigger empty-stop retry or any other auto-continuation. + * Regression: a terminal `yield` must stop the current prompt loop before a + * provider continuation can produce a trailing empty assistant `stop`. * * The session's executor treats a successful yield as the terminal result for - * a scripted subagent run; if the empty-stop recovery path then schedules - * `agent.continue()`, the already-yielded child resumes and produces post-yield - * tool calls (see issue #3389). + * a scripted subagent run; if the loop continues after that tool result, the + * already-yielded child resumes and can enter post-yield retries or tool calls + * (see issues #3389 and #4963). */ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; @@ -147,20 +147,17 @@ afterEach(async () => { }); describe("AgentSession yield empty-stop suppression", () => { - it("does not retry a trailing empty assistant stop after a successful yield", async () => { - const { session, mock } = await createHarness([yieldCall("done", "call-yield-done"), emptyStop()]); + it("does not continue to a trailing empty assistant stop after a successful yield", async () => { + const { session, mock } = await createHarness([yieldCall("done", "call-yield-done")]); await session.prompt("do work then yield"); await session.waitForIdle(); - // Two model calls: the yield turn and the trailing empty stop. Without the - // fix, the empty stop would schedule a `continue()` and either drive a - // third call or throw "no response configured" from the mock. - expect(mock.calls).toHaveLength(2); + expect(mock.calls).toHaveLength(1); expect(reminderMessages(session.agent.state.messages)).toHaveLength(0); }); - it("suppresses multiple trailing empty stops within the same yield-terminated run", async () => { + it("stops at the terminal yield instead of consuming scripted trailing empty stops", async () => { const { session, mock } = await createHarness([ yieldCall("done", "call-yield-multi"), emptyStop(), @@ -171,18 +168,14 @@ describe("AgentSession yield empty-stop suppression", () => { await session.prompt("yield then maybe trail"); await session.waitForIdle(); - // Without suppression, empty-stop retries would consume extra mock entries - // and append at least one reminder. With the fix, the loop ends at the - // first trailing empty stop. - expect(mock.calls).toHaveLength(2); + expect(mock.calls).toHaveLength(1); expect(reminderMessages(session.agent.state.messages)).toHaveLength(0); }); it("clears yield-termination on the next prompt so empty stops retry normally", async () => { const { session, mock } = await createHarness([ - // Run 1: yield then trailing empty stop. Suppression applies. + // Run 1: terminal yield stops without consuming a trailing provider response. yieldCall("first", "call-yield-first"), - emptyStop(), // Run 2: empty stop should retry as usual now that the flag has cleared. recordCall("alpha", "call-record-alpha"), emptyStop(), @@ -191,7 +184,7 @@ describe("AgentSession yield empty-stop suppression", () => { await session.prompt("yield first"); await session.waitForIdle(); - expect(mock.calls).toHaveLength(2); + expect(mock.calls).toHaveLength(1); expect(reminderMessages(session.agent.state.messages)).toHaveLength(0); await session.prompt("now record"); @@ -199,15 +192,14 @@ describe("AgentSession yield empty-stop suppression", () => { // Three additional calls (record, emptyStop, finished). Exactly one // empty-stop reminder injected on the second run. - expect(mock.calls).toHaveLength(5); + expect(mock.calls).toHaveLength(4); expect(reminderMessages(session.agent.state.messages)).toHaveLength(1); }); it("treats an idle IRC wake after a yielded run as a fresh turn for empty-stop retry", async () => { const { session, mock } = await createHarness([ - // Run 1: yield then trailing empty stop. Suppression applies only to this yielded run. + // Run 1: terminal yield stops without consuming a trailing provider response. yieldCall("first", "call-yield-before-irc"), - emptyStop(), // Run 2: an idle IRC wake is a fresh turn, so its empty stop should retry normally. emptyStop(), { content: ["recovered after IRC retry"], stopReason: "stop" }, @@ -215,7 +207,7 @@ describe("AgentSession yield empty-stop suppression", () => { await session.prompt("yield first"); await session.waitForIdle(); - expect(mock.calls).toHaveLength(2); + expect(mock.calls).toHaveLength(1); expect(reminderMessages(session.agent.state.messages)).toHaveLength(0); const outcome = await session.deliverIrcMessage({ @@ -228,7 +220,7 @@ describe("AgentSession yield empty-stop suppression", () => { expect(outcome).toBe("woken"); await session.waitForIdle(); - expect(mock.calls).toHaveLength(4); + expect(mock.calls).toHaveLength(3); expect(reminderMessages(session.agent.state.messages)).toHaveLength(1); expect(assistantText(session.agent.state.messages)).toContain("recovered after IRC retry"); }); From b9e560d806f42c4c2dbdf9d2024e4b383a3781ed Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 21:57:47 +0200 Subject: [PATCH 011/205] fix(ai-registry): migrated xAI auth to device flow for reliability - Implemented standard RFC 8628 device authorization flow for xAI Grok. - Introduced a generic polling utility to manage device code authorization status. - Replaced the previous PKCE-based callback server flow to improve authentication reliability. - Updated authentication logic to handle specific OAuth response scenarios like polling delays and pending authorization. --- packages/ai/CHANGELOG.md | 8 + .../oauth/__tests__/xai-oauth.test.ts | 271 +++++++---- packages/ai/src/registry/oauth/device-code.ts | 92 ++++ packages/ai/src/registry/oauth/index.ts | 92 +--- packages/ai/src/registry/oauth/xai-oauth.ts | 436 ++++++++++-------- packages/ai/src/registry/xai-oauth.ts | 1 - packages/ai/test/provider-registry.test.ts | 1 - 7 files changed, 530 insertions(+), 371 deletions(-) create mode 100644 packages/ai/src/registry/oauth/device-code.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index e6d81932f..f159a1eb6 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Changed + +- Updated xAI OAuth to use a dedicated device-code flow instead of redirect/loopback server + +### Fixed + +- Fixed xAI Grok OAuth login to use xAI's device authorization flow: `/login` now opens the verification URL, displays the device code, and polls for approval instead of asking for a pasted redirect or linking to Hermes Agent documentation. + ## [16.3.14] - 2026-07-09 ### Changed diff --git a/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts b/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts index e80a6a544..ac5012ba0 100644 --- a/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts +++ b/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; -import { isXAIAccessTokenExpiring, refreshXAIOAuthToken, validateXAIEndpoint, XAIOAuthFlow } from "../xai-oauth"; +import { isXAIAccessTokenExpiring, loginXAIOAuth, refreshXAIOAuthToken, validateXAIEndpoint } from "../xai-oauth"; afterEach(() => { vi.restoreAllMocks(); @@ -11,6 +11,76 @@ function jwtWithExp(exp: number): string { return `${header}.${payload}.sig`; } +const DISCOVERY_URL = "https://auth.x.ai/.well-known/openid-configuration"; +const DEVICE_CODE_URL = "https://auth.x.ai/oauth2/device/code"; +const TOKEN_ENDPOINT = "https://auth.x.ai/oauth2/token"; +const CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"; +const SCOPE = "openid profile email offline_access grok-cli:access api:access"; + +const DEVICE_AUTHORIZATION = { + device_code: "device-code-123", + user_code: "ABCD-EFGH", + verification_uri: "https://auth.x.ai/activate", + verification_uri_complete: "https://auth.x.ai/activate?user_code=ABCD-EFGH", + expires_in: 600, + interval: 1, +}; + +type RecordedRequest = { + url: string; + init: RequestInit | undefined; +}; + +type TokenResponse = { + body: unknown; + status?: number; +}; + +function jsonResponse(body: unknown, status: number = 200): Response { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +function createDeviceFlowFetch(tokenResponses: readonly TokenResponse[]) { + const requests: RecordedRequest[] = []; + let tokenResponseIndex = 0; + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(); + requests.push({ url, init }); + + if (url === DISCOVERY_URL) { + return jsonResponse({ token_endpoint: TOKEN_ENDPOINT }); + } + if (url === DEVICE_CODE_URL) { + return jsonResponse(DEVICE_AUTHORIZATION); + } + if (url === TOKEN_ENDPOINT) { + const tokenResponse = tokenResponses[tokenResponseIndex]; + tokenResponseIndex += 1; + if (!tokenResponse) { + throw new Error(`Unexpected xAI token poll ${tokenResponseIndex}`); + } + return jsonResponse(tokenResponse.body, tokenResponse.status); + } + throw new Error(`Unexpected xAI OAuth request: ${url}`); + }); + + return { + fetchMock: fetchMock as unknown as typeof fetch, + requests, + }; +} + +function requestForm(request: RecordedRequest | undefined): URLSearchParams { + const body = request?.init?.body; + if (!(body instanceof URLSearchParams)) { + throw new Error("Expected an application/x-www-form-urlencoded request body"); + } + return body; +} + describe("isXAIAccessTokenExpiring", () => { it("returns false for an empty string", () => { expect(isXAIAccessTokenExpiring("")).toBe(false); @@ -63,97 +133,138 @@ describe("refreshXAIOAuthToken", () => { }); }); -describe("XAIOAuthFlow", () => { - it("pins the redirect URI to xAI's allowlisted loopback port", () => { - const flow = new XAIOAuthFlow({}); - - expect(flow.redirectUri).toBe("http://127.0.0.1:56121/callback"); - }); - - it("uses pasted-code login without starting a callback server", async () => { - const serveSpy = vi.spyOn(Bun, "serve").mockImplementation(() => { - throw new Error("callback server should not start"); - }); - let authUrl = ""; - let tokenRequestBody = ""; - const progress: string[] = []; - const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { - const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(); - if (url.includes("/.well-known/openid-configuration")) { - return new Response( - JSON.stringify({ - authorization_endpoint: "https://auth.x.ai/oauth/authorize", - token_endpoint: "https://auth.x.ai/oauth/token", - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); - } - tokenRequestBody = init?.body instanceof URLSearchParams ? init.body.toString() : String(init?.body ?? ""); - return new Response( - JSON.stringify({ +describe("loginXAIOAuth", () => { + it("performs the RFC 8628 device flow and returns the issued credentials", async () => { + const now = 1_800_000_000_000; + vi.spyOn(Date, "now").mockReturnValue(now); + const { fetchMock, requests } = createDeviceFlowFetch([ + { + body: { access_token: "access-token", refresh_token: "refresh-token", expires_in: 3600, - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); + }, + }, + ]); + const authEvents: Array<{ url: string; instructions?: string }> = []; + const progress: string[] = []; + const onAuth = vi.fn((info: { url: string; instructions?: string }) => { + authEvents.push(info); + }); + const onProgress = vi.fn((message: string) => { + progress.push(message); + }); + const onManualCodeInput = vi.fn(async () => { + throw new Error("device authorization must not request a pasted code"); }); - const flow = new XAIOAuthFlow({ - fetch: fetchMock as unknown as typeof fetch, - onAuth: info => { - authUrl = info.url; - }, - onManualCodeInput: async () => { - const parsed = new URL(authUrl); - const redirectUri = parsed.searchParams.get("redirect_uri") ?? ""; - const state = parsed.searchParams.get("state") ?? ""; - return `${redirectUri}?code=code-xyz&state=${encodeURIComponent(state)}`; - }, - onProgress: message => progress.push(message), + const credentials = await loginXAIOAuth({ + fetch: fetchMock, + onAuth, + onProgress, + onManualCodeInput, }); - const credentials = await flow.login(); - const authorizeUrl = new URL(authUrl); - const tokenParams = new URLSearchParams(tokenRequestBody); + expect(requests.map(request => request.url)).toEqual([DISCOVERY_URL, DEVICE_CODE_URL, TOKEN_ENDPOINT]); - expect(serveSpy).not.toHaveBeenCalled(); - expect(authorizeUrl.searchParams.get("redirect_uri")).toBe("http://127.0.0.1:56121/callback"); - expect(progress).toContain("Waiting for pasted authorization code..."); - expect(tokenParams.get("code")).toBe("code-xyz"); - expect(credentials.access).toBe("access-token"); - expect(credentials.refresh).toBe("refresh-token"); + const discoveryRequest = requests[0]; + expect(discoveryRequest?.init?.method).toBe("GET"); + expect(new Headers(discoveryRequest?.init?.headers).get("Accept")).toBe("application/json"); + + const deviceRequest = requests[1]; + expect(deviceRequest?.init?.method).toBe("POST"); + const deviceHeaders = new Headers(deviceRequest?.init?.headers); + expect(deviceHeaders.get("Content-Type")).toBe("application/x-www-form-urlencoded"); + expect(deviceHeaders.get("Accept")).toBe("application/json"); + const deviceForm = requestForm(deviceRequest); + expect([...deviceForm.keys()].sort()).toEqual(["client_id", "scope"]); + expect(Object.fromEntries(deviceForm)).toEqual({ + client_id: CLIENT_ID, + scope: SCOPE, + }); + + const tokenRequest = requests[2]; + expect(tokenRequest?.init?.method).toBe("POST"); + const tokenHeaders = new Headers(tokenRequest?.init?.headers); + expect(tokenHeaders.get("Content-Type")).toBe("application/x-www-form-urlencoded"); + expect(tokenHeaders.get("Accept")).toBe("application/json"); + const tokenForm = requestForm(tokenRequest); + expect([...tokenForm.keys()].sort()).toEqual(["client_id", "device_code", "grant_type"]); + expect(Object.fromEntries(tokenForm)).toEqual({ + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }); + + expect(authEvents).toEqual([ + { + url: DEVICE_AUTHORIZATION.verification_uri_complete, + instructions: `Enter code: ${DEVICE_AUTHORIZATION.user_code}`, + }, + ]); + expect(authEvents[0]?.instructions).not.toMatch(/hermes/i); + expect(onManualCodeInput).not.toHaveBeenCalled(); + expect(progress).toEqual(["Waiting for xAI device authorization..."]); + expect(credentials).toEqual({ + access: "access-token", + refresh: "refresh-token", + expires: now + 3_300_000, + }); }); -}); -describe("XAIOAuthFlow.exchangeToken", () => { - it("rejects when the token-exchange response is missing access_token", async () => { - const fetchMock = vi.fn(async (input: string | URL) => { - const url = typeof input === "string" ? input : input.toString(); - if (url.includes("/.well-known/openid-configuration")) { - return new Response( - JSON.stringify({ - authorization_endpoint: "https://auth.x.ai/oauth/authorize", - token_endpoint: "https://auth.x.ai/oauth/token", - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); - } - // Token-exchange response deliberately omits `access_token` to exercise - // the missing-token rejection path. The value of `refresh_token` here is - // a literal test marker, not a real secret — the test verifies - // exchangeToken throws before any token would be persisted. - return new Response(JSON.stringify({ refresh_token: "stub-refresh-token-for-test-only" }), { - status: 200, - headers: { "Content-Type": "application/json" }, - }); - }); + it("continues through authorization_pending and slow_down responses", async () => { + const sleepSpy = vi.spyOn(Bun, "sleep").mockResolvedValue(undefined); + const { fetchMock, requests } = createDeviceFlowFetch([ + { status: 400, body: { error: "authorization_pending" } }, + { status: 400, body: { error: "slow_down" } }, + { + body: { + access_token: "eventual-access-token", + refresh_token: "eventual-refresh-token", + expires_in: 3600, + }, + }, + ]); - const flow = new XAIOAuthFlow({ fetch: fetchMock as unknown as typeof fetch }); - await flow.generateAuthUrl("state-abc", "http://127.0.0.1:56121/callback"); + const credentials = await loginXAIOAuth({ fetch: fetchMock }); - await expect(flow.exchangeToken("code-xyz", "state-abc", "http://127.0.0.1:56121/callback")).rejects.toThrow( - /access_token/, + const tokenRequests = requests.filter(request => request.url === TOKEN_ENDPOINT); + expect(tokenRequests).toHaveLength(3); + expect(tokenRequests.map(request => Object.fromEntries(requestForm(request)))).toEqual([ + { + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }, + { + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }, + { + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }, + ]); + expect(sleepSpy.mock.calls).toEqual([[1000], [6000]]); + expect(credentials.access).toBe("eventual-access-token"); + expect(credentials.refresh).toBe("eventual-refresh-token"); + }); + + it("rejects a token response that omits access_token", async () => { + const { fetchMock, requests } = createDeviceFlowFetch([ + { + body: { + refresh_token: "refresh-token", + expires_in: 3600, + }, + }, + ]); + + await expect(loginXAIOAuth({ fetch: fetchMock })).rejects.toThrow( + /xAI device-code token response missing access_token/, ); + expect(requests.filter(request => request.url === TOKEN_ENDPOINT)).toHaveLength(1); }); }); diff --git a/packages/ai/src/registry/oauth/device-code.ts b/packages/ai/src/registry/oauth/device-code.ts new file mode 100644 index 000000000..fc2dd723d --- /dev/null +++ b/packages/ai/src/registry/oauth/device-code.ts @@ -0,0 +1,92 @@ +import * as AIError from "../../error"; + +const DEVICE_FLOW_CANCEL_MESSAGE = "Login cancelled"; +const DEVICE_FLOW_TIMEOUT_MESSAGE = "Device flow timed out"; +const DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE = + "Device flow timed out after one or more slow_down responses. This is often caused by clock drift in WSL or VM environments. Please sync or restart the VM clock and try again."; +const MINIMUM_DEVICE_FLOW_INTERVAL_MS = 1000; +const DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS = 5; +const SLOW_DOWN_INTERVAL_INCREMENT_MS = 5000; + +/** Result returned by one OAuth device-code polling attempt. */ +export type OAuthDeviceCodePollResult = + | { status: "complete"; value: T } + | { status: "pending" } + | { status: "slow_down" } + | { status: "failed"; message: string }; + +/** Options for polling an RFC 8628-style OAuth device-code flow. */ +export interface OAuthDeviceCodeFlowOptions { + /** Poll the provider once and classify the response. */ + poll(): OAuthDeviceCodePollResult | Promise>; + /** Provider-requested polling cadence; defaults to RFC 8628's five seconds. */ + intervalSeconds?: number; + /** Provider-issued expiry window for the device code. */ + expiresInSeconds?: number; + /** Cancels the flow with the legacy "Login cancelled" error. */ + signal?: AbortSignal; +} + +async function abortableDeviceFlowSleep(ms: number, signal: AbortSignal | undefined): Promise { + if (!signal) { + await Bun.sleep(ms); + return; + } + if (signal.aborted) { + throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); + } + + const { promise, resolve, reject } = Promise.withResolvers(); + let timer: Timer | undefined; + const onAbort = () => { + clearTimeout(timer); + reject(new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE)); + }; + timer = setTimeout(() => { + signal.removeEventListener("abort", onAbort); + resolve(); + }, ms); + signal.addEventListener("abort", onAbort, { once: true }); + await promise; +} + +/** Poll an OAuth device-code flow until completion, provider failure, timeout, or cancellation. */ +export async function pollOAuthDeviceCodeFlow(options: OAuthDeviceCodeFlowOptions): Promise { + const deadline = + typeof options.expiresInSeconds === "number" + ? Date.now() + options.expiresInSeconds * 1000 + : Number.POSITIVE_INFINITY; + let intervalMs = Math.max( + MINIMUM_DEVICE_FLOW_INTERVAL_MS, + Math.floor((options.intervalSeconds ?? DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS) * 1000), + ); + let slowDownResponses = 0; + + while (Date.now() < deadline) { + if (options.signal?.aborted) { + throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); + } + const result = await options.poll(); + if (result.status === "complete") { + return result.value; + } + if (result.status === "failed") { + throw new AIError.OAuthError(result.message, { kind: "polling" }); + } + if (result.status === "slow_down") { + slowDownResponses += 1; + intervalMs = Math.max(MINIMUM_DEVICE_FLOW_INTERVAL_MS, intervalMs + SLOW_DOWN_INTERVAL_INCREMENT_MS); + } + + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) { + break; + } + await abortableDeviceFlowSleep(Math.min(intervalMs, remainingMs), options.signal); + } + + throw new AIError.OAuthError( + slowDownResponses > 0 ? DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE : DEVICE_FLOW_TIMEOUT_MESSAGE, + { kind: "timeout" }, + ); +} diff --git a/packages/ai/src/registry/oauth/index.ts b/packages/ai/src/registry/oauth/index.ts index 4cca3a778..31d402ce4 100644 --- a/packages/ai/src/registry/oauth/index.ts +++ b/packages/ai/src/registry/oauth/index.ts @@ -12,99 +12,9 @@ import type { OAuthProviderInterface, } from "./types"; +export * from "./device-code"; export type * from "./types"; -const DEVICE_FLOW_CANCEL_MESSAGE = "Login cancelled"; -const DEVICE_FLOW_TIMEOUT_MESSAGE = "Device flow timed out"; -const DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE = - "Device flow timed out after one or more slow_down responses. This is often caused by clock drift in WSL or VM environments. Please sync or restart the VM clock and try again."; -const MINIMUM_DEVICE_FLOW_INTERVAL_MS = 1000; -const DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS = 5; -const SLOW_DOWN_INTERVAL_INCREMENT_MS = 5000; - -/** Result returned by one OAuth device-code polling attempt. */ -export type OAuthDeviceCodePollResult = - | { status: "complete"; value: T } - | { status: "pending" } - | { status: "slow_down" } - | { status: "failed"; message: string }; - -/** Options for polling an RFC 8628-style OAuth device-code flow. */ -export interface OAuthDeviceCodeFlowOptions { - /** Poll the provider once and classify the response. */ - poll(): OAuthDeviceCodePollResult | Promise>; - /** Provider-requested polling cadence; defaults to RFC 8628's five seconds. */ - intervalSeconds?: number; - /** Provider-issued expiry window for the device code. */ - expiresInSeconds?: number; - /** Cancels the flow with the legacy "Login cancelled" error. */ - signal?: AbortSignal; -} - -async function abortableDeviceFlowSleep(ms: number, signal: AbortSignal | undefined): Promise { - if (!signal) { - await Bun.sleep(ms); - return; - } - if (signal.aborted) { - throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); - } - - const { promise, resolve, reject } = Promise.withResolvers(); - let timer: Timer | undefined; - const onAbort = () => { - if (timer) clearTimeout(timer); - reject(new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE)); - }; - timer = setTimeout(() => { - signal.removeEventListener("abort", onAbort); - resolve(); - }, ms); - signal.addEventListener("abort", onAbort, { once: true }); - await promise; -} - -/** Poll an OAuth device-code flow until completion, provider failure, timeout, or cancellation. */ -export async function pollOAuthDeviceCodeFlow(options: OAuthDeviceCodeFlowOptions): Promise { - const deadline = - typeof options.expiresInSeconds === "number" - ? Date.now() + options.expiresInSeconds * 1000 - : Number.POSITIVE_INFINITY; - let intervalMs = Math.max( - MINIMUM_DEVICE_FLOW_INTERVAL_MS, - Math.floor((options.intervalSeconds ?? DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS) * 1000), - ); - let slowDownResponses = 0; - - while (Date.now() < deadline) { - if (options.signal?.aborted) { - throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); - } - const result = await options.poll(); - if (result.status === "complete") { - return result.value; - } - if (result.status === "failed") { - throw new AIError.OAuthError(result.message, { kind: "polling" }); - } - if (result.status === "slow_down") { - slowDownResponses += 1; - intervalMs = Math.max(MINIMUM_DEVICE_FLOW_INTERVAL_MS, intervalMs + SLOW_DOWN_INTERVAL_INCREMENT_MS); - } - - const remainingMs = deadline - Date.now(); - if (remainingMs <= 0) { - break; - } - await abortableDeviceFlowSleep(Math.min(intervalMs, remainingMs), options.signal); - } - - throw new AIError.OAuthError( - slowDownResponses > 0 ? DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE : DEVICE_FLOW_TIMEOUT_MESSAGE, - { kind: "timeout" }, - ); -} - const builtInOAuthProviders: OAuthProviderInfo[] = PROVIDER_REGISTRY.filter( provider => provider.login && provider.showInLoginList !== false, ).map(provider => ({ diff --git a/packages/ai/src/registry/oauth/xai-oauth.ts b/packages/ai/src/registry/oauth/xai-oauth.ts index 8c55de347..d31da2750 100644 --- a/packages/ai/src/registry/oauth/xai-oauth.ts +++ b/packages/ai/src/registry/oauth/xai-oauth.ts @@ -1,31 +1,22 @@ -// Ported from NousResearch/hermes-agent (MIT) — hermes_cli/auth.py xAI sections (L93-111, L2979-3160, L5286-5469). +// Device authorization and token refresh adapted from NousResearch/hermes-agent (MIT). /** - * xAI Grok (SuperGrok or X Premium+) OAuth flow. + * xAI Grok OAuth device authorization flow. * - * Manual-code PKCE flow using `127.0.0.1:56121/callback` as the allowlisted - * redirect URI. One token unlocks Grok-4.x - * chat, Grok Imagine image generation, and Grok Voice TTS via subsequent - * commits. Endpoint discovery is hardened against MITM via - * {@link validateXAIEndpoint}: any non-HTTPS or non-`x.ai`/`*.x.ai` host is - * rejected on every call site, not just the first. + * Requests an RFC 8628 device code, opens xAI's verification page, and polls + * the discovered token endpoint until the user approves the login. */ import * as AIError from "../../error"; import type { FetchImpl } from "../../types"; -import { OAuthCallbackFlow, type OAuthCallbackFlowOptions } from "./callback-server"; -import { generatePKCE } from "./pkce"; +import { type OAuthDeviceCodePollResult, pollOAuthDeviceCodeFlow } from "./device-code"; import type { OAuthController, OAuthCredentials } from "./types"; -// Hermes hermes_cli/auth.py L93-111 const XAI_OAUTH_ISSUER = "https://auth.x.ai"; const XAI_OAUTH_DISCOVERY_URL = `${XAI_OAUTH_ISSUER}/.well-known/openid-configuration`; +const XAI_OAUTH_DEVICE_CODE_URL = `${XAI_OAUTH_ISSUER}/oauth2/device/code`; const XAI_OAUTH_CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"; const XAI_OAUTH_SCOPE = "openid profile email offline_access grok-cli:access api:access"; -const XAI_OAUTH_REDIRECT_HOST = "127.0.0.1"; -const XAI_OAUTH_REDIRECT_PORT = 56121; -const XAI_OAUTH_REDIRECT_PATH = "/callback"; -const XAI_OAUTH_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/guides/xai-grok-oauth"; // Mirrors the 5-min skew used by anthropic.ts:160 — keeps every provider on the // same conservative client-side expiry window. @@ -35,18 +26,27 @@ const DISCOVERY_TIMEOUT_MS = 15_000; const TOKEN_REQUEST_TIMEOUT_MS = 20_000; interface XAIOAuthDiscovery { - authorization_endpoint: string; token_endpoint: string; } +interface XAIDeviceAuthorization { + deviceCode: string; + userCode: string; + verificationUriComplete: string; + expiresInSeconds: number; + intervalSeconds: number; +} + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null; +} + /** - * Validate an xAI OIDC discovery endpoint against scheme + host. + * Validate an xAI OIDC endpoint against its scheme and host. * - * Hermes `_xai_validate_oauth_endpoint` L2997-3035. The discovery response is - * long-lived and cached in {@link OAuthCredentials}; a single MITM during - * initial login could substitute a malicious `token_endpoint` that would then - * receive every future refresh_token. Rejecting non-HTTPS or non-`x.ai` / - * `*.x.ai` hosts pins the cached endpoint to the xAI auth origin. + * The discovery response is long-lived and its token endpoint receives every + * future refresh token. Rejecting non-HTTPS or non-`x.ai` / `*.x.ai` hosts + * pins that endpoint to the xAI auth origin. * * @throws Error with message `Invalid xAI : ` when the URL fails * either scheme or host validation. @@ -68,11 +68,7 @@ export function validateXAIEndpoint(url: string, field: string): string { return url; } -/** - * Fetch xAI's OIDC discovery document and validate both endpoints. - * - * Hermes `_xai_oauth_discovery` L3038-3084. - */ +/** Fetch xAI's OIDC discovery document and validate the token endpoint. */ async function xaiOAuthDiscovery( timeoutMs: number = DISCOVERY_TIMEOUT_MS, fetchOverride?: FetchImpl, @@ -111,37 +107,29 @@ async function xaiOAuthDiscovery( { kind: "validation", provider: "xai", cause: error }, ); } - if (!payload || typeof payload !== "object") { + if (!isRecord(payload)) { throw new AIError.OAuthError("xAI OIDC discovery response was not a JSON object.", { kind: "validation", provider: "xai", }); } - const obj = payload as Record; - const authorizationEndpoint = - typeof obj.authorization_endpoint === "string" ? obj.authorization_endpoint.trim() : ""; - const tokenEndpoint = typeof obj.token_endpoint === "string" ? obj.token_endpoint.trim() : ""; - if (!authorizationEndpoint || !tokenEndpoint) { - throw new AIError.OAuthError("xAI OIDC discovery response was missing required endpoints.", { + const tokenEndpoint = typeof payload.token_endpoint === "string" ? payload.token_endpoint.trim() : ""; + if (!tokenEndpoint) { + throw new AIError.OAuthError("xAI OIDC discovery response was missing token_endpoint.", { kind: "validation", provider: "xai", }); } - validateXAIEndpoint(authorizationEndpoint, "authorization_endpoint"); validateXAIEndpoint(tokenEndpoint, "token_endpoint"); - return { - authorization_endpoint: authorizationEndpoint, - token_endpoint: tokenEndpoint, - }; + return { token_endpoint: tokenEndpoint }; } /** * Check whether a JWT access token is at or past its `exp` claim (with an * optional refresh-skew margin). * - * Hermes `_xai_access_token_is_expiring` L2979-2994. Returns `false` for any - * malformed input — this is a refresh-trigger check, not a validation, so - * non-JWTs ("no token in cache") must NOT trigger a spurious refresh. + * Returns `false` for malformed input because this is a refresh-trigger check, + * not token validation. */ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): boolean { try { @@ -151,7 +139,8 @@ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): const payloadPart = parts[1]; if (!payloadPart) return false; const decoded = Buffer.from(payloadPart, "base64url").toString("utf8"); - const payload = JSON.parse(decoded) as { exp?: unknown }; + const payload: unknown = JSON.parse(decoded); + if (!isRecord(payload)) return false; const exp = payload.exp; if (typeof exp !== "number" || !Number.isFinite(exp)) return false; const now = Math.floor(Date.now() / 1000); @@ -162,161 +151,232 @@ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): } } -interface BuildXAIAuthorizeUrlOptions { - authorizationEndpoint: string; - redirectUri: string; - codeChallenge: string; - state: string; - nonce: string; -} - -/** - * Build the xAI authorization URL. - * - * Hermes `_xai_oauth_build_authorize_url` L5286-5312. `plan=generic` opts the - * consent screen into xAI's generic OAuth plan tier; without it, - * `accounts.x.ai` rejects loopback OAuth from non-allowlisted clients. - * `referrer=oh-my-pi` lets xAI attribute oh-my-pi-originated logins in their - * OAuth server logs (Hermes uses `referrer=hermes-agent`; oh-my-pi mirrors the - * pattern with its own attribution string). - */ -function buildXAIAuthorizeUrl(opts: BuildXAIAuthorizeUrlOptions): string { - const params = new URLSearchParams({ - response_type: "code", - client_id: XAI_OAUTH_CLIENT_ID, - redirect_uri: opts.redirectUri, - scope: XAI_OAUTH_SCOPE, - code_challenge: opts.codeChallenge, - code_challenge_method: "S256", - state: opts.state, - nonce: opts.nonce, - plan: "generic", - referrer: "oh-my-pi", - }); - return `${opts.authorizationEndpoint}?${params.toString()}`; -} - -/** - * xAI Grok OAuth code flow (Hermes `_xai_oauth_loopback_login` L5315-5469). - */ -export class XAIOAuthFlow extends OAuthCallbackFlow { - #verifier: string = ""; - #fetch: FetchImpl; - - constructor(ctrl: OAuthController) { - super(ctrl, { - preferredPort: XAI_OAUTH_REDIRECT_PORT, - callbackPath: XAI_OAUTH_REDIRECT_PATH, - callbackHostname: XAI_OAUTH_REDIRECT_HOST, - redirectUri: `http://${XAI_OAUTH_REDIRECT_HOST}:${XAI_OAUTH_REDIRECT_PORT}${XAI_OAUTH_REDIRECT_PATH}`, - manualInputOnly: true, - } satisfies OAuthCallbackFlowOptions); - this.#fetch = ctrl.fetch ?? fetch; +function parseXAIDeviceAuthorization(payload: unknown): XAIDeviceAuthorization { + if (!isRecord(payload)) { + throw new AIError.OAuthError("xAI device-code response was not a JSON object.", { + kind: "validation", + provider: "xai", + }); } - async generateAuthUrl(state: string, redirectUri: string): Promise<{ url: string; instructions?: string }> { - const pkce = await generatePKCE(); - this.#verifier = pkce.verifier; - const nonce = crypto.randomUUID().replace(/-/g, ""); - - const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, this.#fetch); - const url = buildXAIAuthorizeUrl({ - authorizationEndpoint: discovery.authorization_endpoint, - redirectUri, - codeChallenge: pkce.challenge, - state, - nonce, + const deviceCode = typeof payload.device_code === "string" ? payload.device_code.trim() : ""; + const userCode = typeof payload.user_code === "string" ? payload.user_code.trim() : ""; + const verificationUri = typeof payload.verification_uri === "string" ? payload.verification_uri.trim() : ""; + const verificationUriComplete = + typeof payload.verification_uri_complete === "string" ? payload.verification_uri_complete.trim() : ""; + const expiresInSeconds = payload.expires_in; + const intervalSeconds = payload.interval; + if ( + !deviceCode || + !userCode || + !verificationUri || + !verificationUriComplete || + typeof expiresInSeconds !== "number" || + !Number.isFinite(expiresInSeconds) || + expiresInSeconds <= 0 || + typeof intervalSeconds !== "number" || + !Number.isFinite(intervalSeconds) || + intervalSeconds <= 0 + ) { + throw new AIError.OAuthError("xAI device-code response missing or invalid required fields.", { + kind: "validation", + provider: "xai", }); - - return { - url, - instructions: `Complete login in your browser for xAI Grok (SuperGrok or X Premium+). Docs: ${XAI_OAUTH_DOCS_URL}`, - }; } - async exchangeToken(code: string, _state: string, redirectUri: string): Promise { - const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, this.#fetch); - const tokenEndpoint = validateXAIEndpoint(discovery.token_endpoint, "token_endpoint"); + validateXAIEndpoint(verificationUri, "verification_uri"); + validateXAIEndpoint(verificationUriComplete, "verification_uri_complete"); + return { + deviceCode, + userCode, + verificationUriComplete, + expiresInSeconds, + intervalSeconds, + }; +} - const body = new URLSearchParams({ - grant_type: "authorization_code", - client_id: XAI_OAUTH_CLIENT_ID, - code, - redirect_uri: redirectUri, - code_verifier: this.#verifier, +function parseXAITokenResponse(payload: unknown, label: string, refreshTokenFallback?: string): OAuthCredentials { + if (!isRecord(payload)) { + throw new AIError.OAuthError(`${label} was not a JSON object`, { + kind: "validation", + provider: "xai", }); + } + const accessToken = typeof payload.access_token === "string" ? payload.access_token : ""; + const responseRefreshToken = typeof payload.refresh_token === "string" ? payload.refresh_token : ""; + const refreshToken = responseRefreshToken || refreshTokenFallback || ""; + const expiresInSeconds = payload.expires_in; + if (!accessToken) { + throw new AIError.OAuthError(`${label} missing access_token`, { + kind: "validation", + provider: "xai", + }); + } + if (!refreshToken) { + throw new AIError.OAuthError(`${label} missing refresh_token`, { + kind: "validation", + provider: "xai", + }); + } + if (typeof expiresInSeconds !== "number" || !Number.isFinite(expiresInSeconds)) { + throw new AIError.OAuthError(`${label} missing expires_in`, { + kind: "validation", + provider: "xai", + }); + } + return { + access: accessToken, + refresh: refreshToken, + expires: Date.now() + expiresInSeconds * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS, + }; +} - const response = await this.#fetch(tokenEndpoint, { +async function requestXAIDeviceAuthorization( + fetchImpl: FetchImpl, + signal?: AbortSignal, +): Promise { + let response: Response; + try { + const timeoutSignal = AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS); + response = await fetchImpl(XAI_OAUTH_DEVICE_CODE_URL, { method: "POST", headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json", }, - body, - signal: AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS), + body: new URLSearchParams({ + client_id: XAI_OAUTH_CLIENT_ID, + scope: XAI_OAUTH_SCOPE, + }), + signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal, }); - - if (!response.ok) { - let detail = ""; - try { - detail = (await response.text()).trim(); - } catch { - // Ignore body-read failures; the status code is the diagnostic. - } - throw new AIError.OAuthError(`xAI token exchange failed: ${response.status}${detail ? ` ${detail}` : ""}`, { - kind: "token-exchange", - provider: "xai", - status: response.status, - }); - } - - let tokenData: { access_token?: unknown; refresh_token?: unknown; expires_in?: unknown }; - try { - tokenData = (await response.json()) as typeof tokenData; - } catch (error) { - throw new AIError.OAuthError( - `xAI token exchange returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`, - { kind: "validation", provider: "xai", cause: error }, - ); - } - - if (typeof tokenData.access_token !== "string" || !tokenData.access_token) { - throw new AIError.OAuthError("xAI token exchange response missing access_token", { - kind: "validation", - provider: "xai", - }); - } - if (typeof tokenData.refresh_token !== "string" || !tokenData.refresh_token) { - throw new AIError.OAuthError("xAI token exchange response missing refresh_token", { - kind: "validation", - provider: "xai", - }); - } - if (typeof tokenData.expires_in !== "number" || !Number.isFinite(tokenData.expires_in)) { - throw new AIError.OAuthError("xAI token exchange response missing expires_in", { - kind: "validation", - provider: "xai", - }); - } - - return { - access: tokenData.access_token, - refresh: tokenData.refresh_token, - expires: Date.now() + tokenData.expires_in * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS, - }; + } catch (error) { + if (signal?.aborted) throw new AIError.LoginCancelledError(); + throw new AIError.OAuthError( + `xAI device-code request failed: ${error instanceof Error ? error.message : String(error)}`, + { kind: "device-auth", provider: "xai", cause: error }, + ); } + + if (!response.ok) { + let detail = ""; + try { + detail = (await response.text()).trim(); + } catch { + // Ignore body-read failures; the status code is the diagnostic. + } + throw new AIError.OAuthError(`xAI device-code request failed: ${response.status}${detail ? ` ${detail}` : ""}`, { + kind: "device-auth", + provider: "xai", + status: response.status, + }); + } + + let payload: unknown; + try { + payload = await response.json(); + } catch (error) { + throw new AIError.OAuthError( + `xAI device-code response returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`, + { kind: "validation", provider: "xai", cause: error }, + ); + } + return parseXAIDeviceAuthorization(payload); } +async function pollXAIDeviceToken( + tokenEndpoint: string, + deviceCode: string, + fetchImpl: FetchImpl, + signal?: AbortSignal, +): Promise> { + let response: Response; + try { + const timeoutSignal = AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS); + response = await fetchImpl(tokenEndpoint, { + method: "POST", + headers: { + "Content-Type": "application/x-www-form-urlencoded", + Accept: "application/json", + }, + body: new URLSearchParams({ + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: XAI_OAUTH_CLIENT_ID, + device_code: deviceCode, + }), + signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal, + }); + } catch (error) { + if (signal?.aborted) throw new AIError.LoginCancelledError(); + throw new AIError.OAuthError( + `xAI device-code token polling failed: ${error instanceof Error ? error.message : String(error)}`, + { kind: "polling", provider: "xai", cause: error }, + ); + } + + let payload: unknown; + try { + payload = await response.json(); + } catch (error) { + throw new AIError.OAuthError( + `xAI device-code token polling returned invalid JSON: ${ + error instanceof Error ? error.message : String(error) + }`, + { kind: "polling", provider: "xai", status: response.status, cause: error }, + ); + } + + if (response.ok) { + return { + status: "complete", + value: parseXAITokenResponse(payload, "xAI device-code token response"), + }; + } + if (!isRecord(payload)) { + throw new AIError.OAuthError(`xAI device-code token polling failed: ${response.status}`, { + kind: "polling", + provider: "xai", + status: response.status, + }); + } + + const errorCode = typeof payload.error === "string" ? payload.error : ""; + if (errorCode === "authorization_pending") return { status: "pending" }; + if (errorCode === "slow_down") return { status: "slow_down" }; + + const errorDescription = typeof payload.error_description === "string" ? payload.error_description : ""; + const detail = errorDescription || errorCode || String(response.status); + throw new AIError.OAuthError(`xAI device-code token polling failed: ${detail}`, { + kind: "polling", + provider: "xai", + status: response.status, + }); +} + +/** Log in to xAI Grok with the RFC 8628 device authorization grant. */ export async function loginXAIOAuth(ctrl: OAuthController): Promise { - return new XAIOAuthFlow(ctrl).login(); + const fetchImpl = ctrl.fetch ?? fetch; + const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, fetchImpl); + const device = await requestXAIDeviceAuthorization(fetchImpl, ctrl.signal); + ctrl.onAuth?.({ + url: device.verificationUriComplete, + instructions: `Enter code: ${device.userCode}`, + }); + ctrl.onProgress?.("Waiting for xAI device authorization..."); + + return pollOAuthDeviceCodeFlow({ + poll: () => pollXAIDeviceToken(discovery.token_endpoint, device.deviceCode, fetchImpl, ctrl.signal), + intervalSeconds: device.intervalSeconds, + expiresInSeconds: device.expiresInSeconds, + signal: ctrl.signal, + }); } /** * Refresh an xAI OAuth access token using a stored refresh_token. * - * Hermes `refresh_xai_oauth_pure` L3087-3160. Re-runs OIDC discovery and - * re-validates the cached `token_endpoint` on the refresh hot path so a - * cached-but-poisoned endpoint cannot silently leak a refresh_token. + * Re-runs OIDC discovery and re-validates the token endpoint before sending + * the stored refresh token. */ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: FetchImpl): Promise { const fetchImpl = fetchOverride ?? fetch; @@ -357,34 +417,14 @@ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: }); } - let data: { access_token?: unknown; refresh_token?: unknown; expires_in?: unknown }; + let payload: unknown; try { - data = (await response.json()) as typeof data; + payload = await response.json(); } catch (error) { throw new AIError.OAuthError( `xAI token refresh returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`, { kind: "validation", provider: "xai", cause: error }, ); } - - if (typeof data.access_token !== "string" || !data.access_token) { - throw new AIError.OAuthError("xAI token refresh response missing access_token", { - kind: "validation", - provider: "xai", - }); - } - if (typeof data.expires_in !== "number" || !Number.isFinite(data.expires_in)) { - throw new AIError.OAuthError("xAI token refresh response missing expires_in", { - kind: "validation", - provider: "xai", - }); - } - - const newRefresh = typeof data.refresh_token === "string" && data.refresh_token ? data.refresh_token : refreshToken; - - return { - access: data.access_token, - refresh: newRefresh, - expires: Date.now() + data.expires_in * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS, - }; + return parseXAITokenResponse(payload, "xAI token refresh response", refreshToken); } diff --git a/packages/ai/src/registry/xai-oauth.ts b/packages/ai/src/registry/xai-oauth.ts index ed1a22bd4..67f1b1bd9 100644 --- a/packages/ai/src/registry/xai-oauth.ts +++ b/packages/ai/src/registry/xai-oauth.ts @@ -14,5 +14,4 @@ export const xaiOauthProvider = { const { refreshXAIOAuthToken } = await import("./oauth/xai-oauth"); return refreshXAIOAuthToken(credentials.refresh); }, - pasteCodeFlow: true, } as const satisfies ProviderDefinition; diff --git a/packages/ai/test/provider-registry.test.ts b/packages/ai/test/provider-registry.test.ts index a72eb59c7..2ed3d88bd 100644 --- a/packages/ai/test/provider-registry.test.ts +++ b/packages/ai/test/provider-registry.test.ts @@ -81,7 +81,6 @@ describe("provider registry auth surface", () => { "google-antigravity", "google-gemini-cli", "openai-codex", - "xai-oauth", ].sort(), ); expect(PASTE_CODE_LOGIN_PROVIDERS.has("zenmux")).toBe(false); From faa70100eaaccf999d3f141ea49f37124c17cd3f Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:20:32 +0200 Subject: [PATCH 012/205] feat: enabled openai reasoning mode and integrated new model catalog - Enabled OpenAI pro reasoning mode by integrating reasoning aliases and parameter injection. - Expanded the model catalog with GPT-5.6 Luna, Sol, Terra, and Meta Muse Spark 1.1. - Updated model type definitions and provider request transformers to support reasoning configurations. - Refined model generation scripts to include new pro-reasoning aliases for OpenAI providers. --- packages/ai/CHANGELOG.md | 4 + .../src/providers/openai-codex-responses.ts | 2 +- .../openai-codex/request-transformer.ts | 8 + .../ai/src/providers/openai-responses-wire.ts | 7 + packages/ai/src/providers/openai-responses.ts | 7 + packages/catalog/CHANGELOG.md | 8 + packages/catalog/scripts/generate-models.ts | 5 + packages/catalog/src/models.json | 733 +++++++++++++++++- .../src/provider-models/openai-compat.ts | 56 ++ packages/catalog/src/types.ts | 7 + 10 files changed, 813 insertions(+), 24 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f159a1eb6..d15331c39 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added OpenAI pro reasoning mode support: models carrying the catalog `reasoningMode: "pro"` marker (GPT-5.6 Pro aliases) send `reasoning: { mode: "pro" }` on OpenAI Responses and Codex Responses requests, alongside the configured effort. The Codex request body now honors `requestModelId` so catalog aliases request the base upstream model id. + ### Changed - Updated xAI OAuth to use a dedicated device-code flow instead of redirect/loopback server diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 0bde541d0..05654f0e0 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -949,7 +949,7 @@ export async function buildTransformedCodexRequestBody( promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), ): Promise { const params: RequestBody = { - model: model.id, + model: model.requestModelId ?? model.id, input: convertMessages(model, context), stream: true, prompt_cache_key: promptCacheKey, diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index 67af8f232..968226016 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -23,6 +23,8 @@ export interface ReasoningConfig { effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; summary?: "auto" | "concise" | "detailed"; context?: CodexReasoningContext; + /** Pro reasoning serving mode (gpt-5.6+ catalog pro aliases). */ + mode?: "pro"; } export interface CodexRequestOptions { @@ -367,6 +369,12 @@ export async function transformRequestBody( } else { delete body.reasoning; } + // Catalog pro aliases (`gpt-5.6-*-pro`): applied after the effort branch so + // the mode is sent even when no effort is set (the branch above deletes + // `body.reasoning` in that case) — mode and effort are independent fields. + if (model.reasoningMode) { + body.reasoning = { ...body.reasoning, mode: model.reasoningMode }; + } body.text = { ...body.text, diff --git a/packages/ai/src/providers/openai-responses-wire.ts b/packages/ai/src/providers/openai-responses-wire.ts index 5246b5eaf..7992ec319 100644 --- a/packages/ai/src/providers/openai-responses-wire.ts +++ b/packages/ai/src/providers/openai-responses-wire.ts @@ -6318,6 +6318,13 @@ export interface Reasoning { * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. */ effort?: ReasoningEffort | null; + /** + * **gpt-5.6 and later models only** + * + * Reasoning serving mode. `pro` routes the request to the pro reasoning + * path (more compute per response); omit for the standard path. + */ + mode?: "pro" | null; /** * @deprecated **Deprecated:** use `summary` instead. * diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 759fdbfdc..ca0f89aa9 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -904,6 +904,13 @@ export function buildParams( model.thinking?.effortMap?.[effort as NonNullable] ?? effort, }); + // Catalog pro aliases (`gpt-5.6-*-pro`): merge AFTER the compat policy so the + // mode survives every policy branch (disabled/omitted effort included) while + // keeping whatever effort/summary the policy produced — mode and effort are + // independent wire fields. + if (model.reasoningMode) { + params.reasoning = { ...params.reasoning, mode: model.reasoningMode }; + } applyOpenAIGatewayRouting(params, model.compat); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index bece2a3cd..937e3ed7f 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Added + +- Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants +- Added `meta/muse-spark-1.1` model support +- Added support for thinking modes on `poolside/laguna` models + +- Added generated GPT-5.6 Pro aliases (`gpt-5.6-{luna,sol,terra}-pro`) on the `openai` and `openai-codex` providers: each alias sends the base model id on the wire (`requestModelId`) with the new `reasoningMode: "pro"` marker, and re-derives from the current base rows on every catalog regeneration. + ## [16.3.14] - 2026-07-09 ### Added diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 09b9548a9..1d9df8b78 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -38,6 +38,7 @@ import { isKimiK27CodeModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, + projectOpenAIProReasoningAliases, SAKANA_FUGU_STATIC_MODELS, stripFireworksDeepSeekThinkingToggle, } from "../src/provider-models/openai-compat"; @@ -585,6 +586,10 @@ async function generateModels() { const name = cleanModelName(model.name); return name === model.name ? model : { ...model, name }; }); + // Re-derive the first-party gpt-5.6 pro-reasoning aliases from the current + // base rows (stale previous-snapshot aliases are dropped inside), before the + // policy re-bake so the aliases get the same baked thinking metadata. + allModels = projectOpenAIProReasoningAliases(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); // Collapse effort-tier variants AFTER the policy re-bake: live-discovery diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 55d87779c..67ed8c8e0 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -32288,7 +32288,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -32299,7 +32299,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-m.1:free": { "id": "poolside/laguna-m.1:free", @@ -32374,7 +32384,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -32385,7 +32395,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs.2:free": { "id": "poolside/laguna-xs.2:free", @@ -45679,6 +45699,25 @@ "contextWindow": 328000, "maxTokens": 65536 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "meta/muse-spark-1.1", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576 + }, "microsoft/MAI-DS-R1-FP8": { "id": "microsoft/MAI-DS-R1-FP8", "name": "microsoft/MAI-DS-R1-FP8", @@ -48497,6 +48536,174 @@ }, "contextPromotionTarget": "nanogpt/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, "openai/gpt-chat-latest": { "id": "openai/gpt-chat-latest", "name": "GPT Chat Latest", @@ -49105,41 +49312,61 @@ }, "poolside/laguna-m.1": { "id": "poolside/laguna-m.1", - "name": "poolside/laguna-m.1", + "name": "Laguna M.1", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.2, + "output": 0.4, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", - "name": "poolside/laguna-xs.2", + "name": "Laguna XS.2", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.2, + "output": 0.4, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "qvq-max": { "id": "qvq-max", @@ -60096,6 +60323,44 @@ }, "contextPromotionTarget": "openai/gpt-5.4" }, + "gpt-5.6": { + "id": "gpt-5.6", + "name": "GPT-5.6", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", @@ -60134,6 +60399,46 @@ } } }, + "gpt-5.6-luna-pro": { + "id": "gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", @@ -60172,6 +60477,46 @@ } } }, + "gpt-5.6-sol-pro": { + "id": "gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", @@ -60210,6 +60555,46 @@ } } }, + "gpt-5.6-terra-pro": { + "id": "gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "o1": { "id": "o1", "name": "o1", @@ -61034,6 +61419,53 @@ } } }, + "gpt-5.6-luna-pro": { + "id": "gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 3, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", @@ -61079,6 +61511,53 @@ } } }, + "gpt-5.6-sol-pro": { + "id": "gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 1, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", @@ -61123,6 +61602,53 @@ "xhigh": "max" } } + }, + "gpt-5.6-terra-pro": { + "id": "gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 2, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } } }, "opencode": { @@ -64155,7 +64681,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -67983,8 +68509,8 @@ ], "cost": { "input": 0.72, - "output": 3.5, - "cacheRead": 0.15, + "output": 3.49, + "cacheRead": 0.159, "cacheWrite": 0 }, "contextWindow": 262144, @@ -69547,7 +70073,7 @@ "input": 1, "output": 6, "cacheRead": 0.09999999999999999, - "cacheWrite": 0 + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69584,7 +70110,7 @@ "input": 1, "output": 6, "cacheRead": 0.09999999999999999, - "cacheWrite": 0 + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69621,7 +70147,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69658,7 +70184,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69695,7 +70221,7 @@ "input": 2.5, "output": 15, "cacheRead": 0.25, - "cacheWrite": 0 + "cacheWrite": 3.125 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69732,7 +70258,7 @@ "input": 2.5, "output": 15, "cacheRead": 0.25, - "cacheWrite": 0 + "cacheWrite": 3.125 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -77243,6 +77769,138 @@ ] } }, + "openai-gpt-56-luna": { + "id": "openai-gpt-56-luna", + "name": "openai-gpt-56-luna", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-luna-pro": { + "id": "openai-gpt-56-luna-pro", + "name": "openai-gpt-56-luna-pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-sol": { + "id": "openai-gpt-56-sol", + "name": "openai-gpt-56-sol", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-sol-pro": { + "id": "openai-gpt-56-sol-pro", + "name": "openai-gpt-56-sol-pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-terra": { + "id": "openai-gpt-56-terra", + "name": "openai-gpt-56-terra", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-terra-pro": { + "id": "openai-gpt-56-terra-pro", + "name": "openai-gpt-56-terra-pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, "openai-gpt-oss-120b": { "id": "openai-gpt-oss-120b", "name": "OpenAI GPT OSS 120B", @@ -80352,6 +81010,35 @@ "contextWindow": 128000, "maxTokens": 8192 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "Muse Spark 1.1", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.25, + "output": 4.25, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax M2", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 83cd58136..1877fa7d2 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -821,6 +821,62 @@ export function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): Mo }; } +/** First-party gpt-5.6 SKUs that accept `reasoning: { mode: "pro" }` on the Responses APIs. */ +const OPENAI_PRO_REASONING_BASE_IDS: Record = { + "gpt-5.6-luna": true, + "gpt-5.6-sol": true, + "gpt-5.6-terra": true, +}; +const OPENAI_PRO_REASONING_PROVIDERS: Record = { openai: true, "openai-codex": true }; + +/** + * A row this generator pass owns: one of the derived `gpt-5.6-*-pro` alias ids + * on `openai`/`openai-codex` that carries the generated `reasoningMode` marker. + * A real upstream model occupying the same id has no `reasoningMode` and is + * never touched. + */ +function isGeneratedOpenAIProReasoningAlias(model: ModelSpec): boolean { + return ( + OPENAI_PRO_REASONING_PROVIDERS[model.provider] === true && + model.reasoningMode !== undefined && + model.id.endsWith("-pro") && + OPENAI_PRO_REASONING_BASE_IDS[model.id.slice(0, -"-pro".length)] === true + ); +} + +/** + * Re-derive the generated pro-reasoning aliases (`gpt-5.6-*-pro`) for the + * first-party `openai`/`openai-codex` gpt-5.6 rows. Each alias inherits the + * base row's metadata, requests the base wire id via `requestModelId`, and + * sets `reasoningMode: "pro"` so Responses-family request builders emit + * `reasoning: { mode: "pro" }`. Called by the models.json generator after all + * sources merge: stale copies of the owned aliases (previous snapshot) are + * dropped and re-projected from the current base rows so alias metadata always + * tracks the base, while a real upstream model that occupies an alias id wins + * and suppresses the projection. + */ +export function projectOpenAIProReasoningAliases(models: readonly ModelSpec[]): ModelSpec[] { + const kept = models.filter(model => !isGeneratedOpenAIProReasoningAlias(model)); + const ids = new Set(kept.map(model => `${model.provider}/${model.id}`)); + const out = [...kept]; + for (const model of kept) { + if (!OPENAI_PRO_REASONING_PROVIDERS[model.provider]) continue; + if (!OPENAI_PRO_REASONING_BASE_IDS[model.id]) continue; + const aliasId = `${model.id}-pro`; + const aliasKey = `${model.provider}/${aliasId}`; + if (ids.has(aliasKey)) continue; + ids.add(aliasKey); + out.push({ + ...model, + id: aliasId, + name: `${model.name} Pro`, + requestModelId: model.id, + reasoningMode: "pro", + }); + } + return out; +} + // --------------------------------------------------------------------------- // 2. Groq // --------------------------------------------------------------------------- diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 103549d0f..c0f4d7620 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -691,6 +691,13 @@ export interface Model { * everything local (selection, caching, usage attribution) keys on `id`. */ requestModelId?: string; + /** + * `reasoning.mode` to send on OpenAI Responses-family requests. Set on + * generated pro aliases (`gpt-5.6-*-pro` on `openai`/`openai-codex`) that + * pair a base wire id (`requestModelId`) with OpenAI's pro reasoning + * serving path. Absent everywhere else; providers omit the wire field. + */ + reasoningMode?: "pro"; name: string; api: TApi; provider: Provider; From fde4a19c6226168c78b989dba1f7c0cfc3fca15d Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:28:58 +0200 Subject: [PATCH 013/205] feat: added prompt-cache affinity support for grok models - Introduced `getOpenAIPromptCacheKey` to provide a unified identity resolution for both cache keys and affinity headers. - Enabled `x-grok-conv-id` header support in the OpenAI completions provider for models configured with cache affinity. - Added comprehensive tests to verify cache affinity header behavior across varied session and cache configuration states. --- .../src/compaction/compaction-v2-streaming.ts | 6 +- packages/ai/CHANGELOG.md | 7 ++ packages/ai/src/auth-gateway/server.ts | 4 +- .../src/providers/azure-openai-responses.ts | 4 +- .../src/providers/openai-codex-responses.ts | 12 +-- .../ai/src/providers/openai-completions.ts | 10 +- packages/ai/src/providers/openai-responses.ts | 6 +- packages/ai/src/providers/openai-shared.ts | 31 +++++-- packages/ai/src/types.ts | 6 +- .../openai-completions-cache-affinity.test.ts | 91 +++++++++++++++++++ packages/catalog/CHANGELOG.md | 8 +- packages/catalog/src/compat/openai.ts | 2 +- 12 files changed, 155 insertions(+), 32 deletions(-) create mode 100644 packages/ai/test/openai-completions-cache-affinity.test.ts diff --git a/packages/agent/src/compaction/compaction-v2-streaming.ts b/packages/agent/src/compaction/compaction-v2-streaming.ts index 4db13bd22..92ecfcdb0 100644 --- a/packages/agent/src/compaction/compaction-v2-streaming.ts +++ b/packages/agent/src/compaction/compaction-v2-streaming.ts @@ -10,7 +10,7 @@ import type { Api, FetchImpl, Model } from "@oh-my-pi/pi-ai"; import { isTransientStatus, ProviderHttpError } from "@oh-my-pi/pi-ai/error"; import { - getOpenAIResponsesPromptCacheKey, + getOpenAIPromptCacheKey, getOpenAIResponsesRoutingSessionId, parseAzureDeploymentNameMap, resolveOpenAIRequestSetup, @@ -269,7 +269,7 @@ async function attemptCompactionV2Streaming( // of an otherwise-normal Responses request, then stream the result. `store` // stays false — compaction must never persist a server-side response object. const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey }; - const promptCacheKey = getOpenAIResponsesPromptCacheKey(cacheOptions); + const promptCacheKey = getOpenAIPromptCacheKey(cacheOptions); const body: Record = { model: request.model, input: [...request.input, COMPACTION_TRIGGER_ITEM], @@ -311,7 +311,7 @@ function buildCompactionV2Headers(model: Model, apiKey: string, request: Compact const api = compactionV2Api(model); const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey }; const routingSessionId = getOpenAIResponsesRoutingSessionId(cacheOptions); - const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(cacheOptions); + const promptCacheSessionId = getOpenAIPromptCacheKey(cacheOptions); const headers: Record = api === "azure-openai-responses" ? { diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d15331c39..96c28897b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,8 +2,15 @@ ## [Unreleased] +### Breaking Changes + +- Renamed `OpenAIResponsesCacheOptions`, `normalizeOpenAIResponsesPromptCacheKey`, and `getOpenAIResponsesPromptCacheKey` to the endpoint-neutral `OpenAICacheOptions`, `normalizeOpenAIPromptCacheKey`, and `getOpenAIPromptCacheKey`. + ### Added +- Added automatic prompt-cache affinity header injection for OpenAI-family chat completions + +- Added support for explicit prompt-cache affinity headers in OpenAI-family chat completions - Added OpenAI pro reasoning mode support: models carrying the catalog `reasoningMode: "pro"` marker (GPT-5.6 Pro aliases) send `reasoning: { mode: "pro" }` on OpenAI Responses and Codex Responses requests, alongside the configured effort. The Codex request body now honors `requestModelId` so catalog aliases request the base upstream model id. ### Changed diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index c7eff02f4..ae4d47415 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -112,8 +112,8 @@ function deriveSessionId(modelId: string, context: Context): string { parts.push(JSON.stringify({ role: first.role, content: first.content })); } const seed = parts.join("\u0000"); - // The 36-char UUID flows through unchanged: Codex's - // `normalizeOpenAIResponsesPromptCacheKey` accepts ≤64 chars verbatim. + // The 36-char UUID flows through unchanged: + // `normalizeOpenAIPromptCacheKey` accepts ≤64 chars verbatim. return deterministicUuid(seed); } diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 276614b2f..31090702c 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -34,7 +34,7 @@ import { applyResponsesReasoningParams, buildResponsesInput, createInitialResponsesAssistantMessage, - getOpenAIResponsesPromptCacheKey, + getOpenAIPromptCacheKey, isOpenAIResponsesProgressEvent, parseAzureDeploymentNameMap, processResponsesStream, @@ -348,7 +348,7 @@ function buildParams( model: deploymentName, input: messages, stream: true, - prompt_cache_key: getOpenAIResponsesPromptCacheKey(options), + prompt_cache_key: getOpenAIPromptCacheKey(options), // Encrypted reasoning replay (applyResponsesReasoningParams) requires // stateless responses, matching the openai provider. store: false, diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 05654f0e0..38bb777a2 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -99,7 +99,7 @@ import { finalizeToolCallArgumentsDone, isOpenAIResponsesProgressEvent, mapOpenAIResponsesStopReason, - normalizeOpenAIResponsesPromptCacheKey, + normalizeOpenAIPromptCacheKey, populateResponsesUsageFromResponse, promoteResponsesToolUseStopReason, } from "./openai-shared"; @@ -898,8 +898,8 @@ async function buildCodexRequestContext( const accountId = getCodexAccountId(apiKey); const baseUrl = model.baseUrl || CODEX_BASE_URL; const url = resolveCodexResponsesUrl(baseUrl); - const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); - const transportSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + const promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); + const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const transformedBody = await buildTransformedCodexRequestBody(model, context, options, promptCacheKey); const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) }; @@ -946,7 +946,7 @@ export async function buildTransformedCodexRequestBody( model: Model<"openai-codex-responses">, context: Context, options: OpenAICodexResponsesOptions | undefined, - promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), + promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), ): Promise { const params: RequestBody = { model: model.requestModelId ?? model.id, @@ -2139,7 +2139,7 @@ export async function prewarmOpenAICodexResponses( const accountId = getCodexAccountId(apiKey); const baseUrl = model.baseUrl || CODEX_BASE_URL; const url = resolveCodexResponsesUrl(baseUrl); - const transportSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const promptCacheKey = transportSessionId; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); const responsesLite = options?.responsesLite === true; @@ -2283,7 +2283,7 @@ function getCodexWebSocketStateForPublicSession( ): CodexWebSocketSessionState | undefined { const baseUrl = options?.baseUrl || model.baseUrl || CODEX_BASE_URL; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); - const normalizedSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + const normalizedSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const publicSessionKey = normalizedSessionId ? `${baseUrl}:${model.id}:${normalizedSessionId}` : undefined; const privateSessionKey = publicSessionKey ? providerSessionState?.webSocketPublicToPrivate.get(publicSessionKey) diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 8006628c2..45a56d9ce 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -82,6 +82,7 @@ import { createInitialResponsesAssistantMessage, createOpenAIStrictToolsState, disableStrictToolsForScope, + getOpenAIPromptCacheKey, getOpenAIStrictToolsScope, isCompiledGrammarTooLargeStrictError, isOpenRouterAnthropicModel, @@ -616,6 +617,7 @@ const streamOpenAICompletionsOnce = ( apiKey, options?.headers, options?.initiatorOverride, + getOpenAIPromptCacheKey(options), ); const premiumRequestsTotal = copilotPremiumRequests; let appliedStrictTools = false; @@ -1359,6 +1361,7 @@ function createRequestSetup( apiKey?: string, extraHeaders?: Record, initiatorOverride?: MessageAttribution, + promptCacheSessionId?: string, ): OpenAIRequestSetup & { baseUrl: string } { const apiVersion = $env.AZURE_OPENAI_API_VERSION || "2024-10-21"; const deploymentName = parseAzureDeploymentNameMap($env.AZURE_OPENAI_DEPLOYMENT_NAME_MAP).get(model.id) ?? model.id; @@ -1366,6 +1369,7 @@ function createRequestSetup( apiKey, extraHeaders, initiatorOverride, + promptCacheSessionId, messages: context.messages, defaultBaseUrl: "https://api.openai.com/v1", // Provider auth/header overlay: Kimi-code hosts require shared client @@ -1413,7 +1417,11 @@ function buildParams( context: Context, options: OpenAICompletionsOptions | undefined, toolStrictModeOverride?: ToolStrictModeOverride, -): { params: OpenAICompletionsParams; toolStrictMode: AppliedToolStrictMode; strictToolsApplied: boolean } { +): { + params: OpenAICompletionsParams; + toolStrictMode: AppliedToolStrictMode; + strictToolsApplied: boolean; +} { const initialPolicy = resolveOpenAICompatForRequest(model, options); const initialCompat = initialPolicy.compat as ResolvedOpenAICompat; diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index ca0f89aa9..3aa740236 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -76,7 +76,7 @@ import { createInitialResponsesAssistantMessage, createOpenAIStrictToolsState, disableStrictToolsForScope, - getOpenAIResponsesPromptCacheKey, + getOpenAIPromptCacheKey, getOpenAIResponsesRoutingSessionId, getOpenAIStrictToolsScope, getOpenRouterResponsesSessionId, @@ -390,7 +390,7 @@ const streamOpenAIResponsesOnce = ( // stable prompt-cache key independently. Side-channel calls use this to // avoid perturbing provider conversation state without cold-starting the cache. const routingSessionId = getOpenAIResponsesRoutingSessionId(options); - const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(options); + const promptCacheSessionId = getOpenAIPromptCacheKey(options); const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; const { headers, copilotPremiumRequests, baseUrl } = resolveOpenAIRequestSetup(model, { apiKey, @@ -818,7 +818,7 @@ export function buildParams( } const cacheRetention = resolveCacheRetention(options?.cacheRetention); - const promptCacheKey = getOpenAIResponsesPromptCacheKey(options); + const promptCacheKey = getOpenAIPromptCacheKey(options); const modelId = applyWireModelIdTransform( model.requestModelId ?? model.id, model.compat.wireModelIdMode, diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index bacf6d8af..c87069ad5 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -123,7 +123,8 @@ export interface OpenAIRequestSetupModel extends OpenAIModelIdentity { compat?: Pick; } -export interface OpenAIResponsesCacheOptions { +/** Cache identity controls shared by OpenAI-family transports. */ +export interface OpenAICacheOptions { cacheRetention?: CacheRetention; sessionId?: string; promptCacheKey?: string; @@ -175,6 +176,14 @@ function applyCoreWeaveProjectHeader(headers: Record): void { } } +function setHeaderIfAbsent(headers: Record, name: string, value: string): void { + const normalizedName = name.toLowerCase(); + for (const existingName in headers) { + if (existingName.toLowerCase() === normalizedName) return; + } + headers[name] = value; +} + export function resolveOpenAIRequestSetup( model: OpenAIRequestSetupModel, options: OpenAIRequestSetupOptions, @@ -257,11 +266,11 @@ export function resolveOpenAIRequestSetup( } if (options.openAISessionId && model.provider === "openai") { - headers.session_id ??= options.openAISessionId; - headers["x-client-request-id"] ??= options.openAISessionId; + setHeaderIfAbsent(headers, "session_id", options.openAISessionId); + setHeaderIfAbsent(headers, "x-client-request-id", options.openAISessionId); } if (options.promptCacheSessionId && model.compat?.promptCacheSessionHeader) { - headers[model.compat.promptCacheSessionHeader] ??= options.promptCacheSessionId; + setHeaderIfAbsent(headers, model.compat.promptCacheSessionHeader, options.promptCacheSessionId); } if (options.defaultBaseUrl !== undefined) { @@ -368,7 +377,8 @@ export function calculateOpenAIUsageAccounting(accounting: OpenAIUsageAccounting }; } -export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined { +/** Normalize a cache identity to the wire limit accepted by OpenAI-family providers. */ +export function normalizeOpenAIPromptCacheKey(sessionId: string | undefined): string | undefined { return normalizeOpenAIStableId(sessionId, 64, "pc_"); } @@ -376,20 +386,21 @@ export function normalizeOpenRouterResponsesSessionId(sessionId: string | undefi return normalizeOpenAIStableId(sessionId, 256, "session_"); } -export function getOpenAIResponsesPromptCacheKey(options: OpenAIResponsesCacheOptions | undefined): string | undefined { +/** Resolve a prompt-cache identity, falling back to the provider session unless caching is disabled. */ +export function getOpenAIPromptCacheKey(options: OpenAICacheOptions | undefined): string | undefined { if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined; - return normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); + return normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); } export function getOpenAIResponsesRoutingSessionId( - options: Pick | undefined, + options: Pick | undefined, ): string | undefined { if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined; - return normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + return normalizeOpenAIPromptCacheKey(options?.sessionId); } export function getOpenRouterResponsesSessionId( - options: Pick | undefined, + options: Pick | undefined, ): string | undefined { if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined; return normalizeOpenRouterResponsesSessionId(options?.sessionId); diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 995fd454f..e94884553 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -388,9 +388,9 @@ export interface StreamOptions { */ sessionId?: string; /** - * Optional prompt-cache identity. When set, OpenAI Responses-compatible - * providers use this for `prompt_cache_key` while keeping `sessionId` for - * provider routing / conversation headers. + * Optional prompt-cache identity. OpenAI-family providers use this for + * `prompt_cache_key` payloads and cache-affinity headers such as + * `x-grok-conv-id`; when omitted, they fall back to `sessionId`. */ promptCacheKey?: string; /** diff --git a/packages/ai/test/openai-completions-cache-affinity.test.ts b/packages/ai/test/openai-completions-cache-affinity.test.ts new file mode 100644 index 000000000..564e22fae --- /dev/null +++ b/packages/ai/test/openai-completions-cache-affinity.test.ts @@ -0,0 +1,91 @@ +import { describe, expect, it } from "bun:test"; +import { type OpenAICompletionsOptions, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; + +const model = getBundledModel<"openai-completions">("xai", "grok-code-fast-1"); +if (!model) throw new Error("Expected bundled xAI Grok model"); +if (model.api !== "openai-completions") throw new Error(`Expected Chat Completions model, received ${model.api}`); +const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }] }; + +function chatCompletionsSse(): Response { + const chunk = (delta: unknown, finishReason: string | null) => + JSON.stringify({ + id: "chatcmpl-affinity", + object: "chat.completion.chunk", + created: 0, + model: model.id, + choices: [{ index: 0, delta, finish_reason: finishReason }], + }); + + return new Response( + `data: ${chunk({ role: "assistant", content: "ok" }, null)}\n\ndata: ${chunk({}, "stop")}\n\ndata: [DONE]\n\n`, + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); +} + +async function captureRequestHeaders(options: OpenAICompletionsOptions): Promise { + let requestHeaders: Headers | undefined; + const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => { + const request = + input instanceof Request + ? new Request(input, init) + : new Request(input instanceof URL ? input.href : input, init); + requestHeaders = request.headers; + return chatCompletionsSse(); + }; + + await streamOpenAICompletions(model, context, { + apiKey: "test-key", + ...options, + fetch: fetchMock, + }).result(); + + if (!requestHeaders) throw new Error("Expected a serialized Chat Completions request"); + return requestHeaders; +} + +describe("openai-completions xAI cache affinity", () => { + const cases: Array<{ + name: string; + options: OpenAICompletionsOptions; + expectedHeader: string | null; + }> = [ + { + name: "uses sessionId when no prompt cache key is provided", + options: { sessionId: "session-fallback" }, + expectedHeader: "session-fallback", + }, + { + name: "keeps the prompt cache key stable across a distinct side-channel session", + options: { promptCacheKey: "stable-cache-key", sessionId: "side-channel-session" }, + expectedHeader: "stable-cache-key", + }, + { + name: "omits automatic affinity when caching is disabled", + options: { + promptCacheKey: "disabled-cache-key", + sessionId: "disabled-session", + cacheRetention: "none", + }, + expectedHeader: null, + }, + { + name: "preserves a caller-provided mixed-case affinity header", + options: { + promptCacheKey: "automatic-cache-key", + sessionId: "automatic-session", + headers: { "X-Grok-Conv-Id": "caller-affinity" }, + }, + expectedHeader: "caller-affinity", + }, + ]; + + for (const { name, options, expectedHeader } of cases) { + it(name, async () => { + const headers = await captureRequestHeaders(options); + + expect(headers.get("x-grok-conv-id")).toBe(expectedHeader); + }); + } +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 937e3ed7f..144a11730 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -7,9 +7,15 @@ - Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants - Added `meta/muse-spark-1.1` model support - Added support for thinking modes on `poolside/laguna` models - - Added generated GPT-5.6 Pro aliases (`gpt-5.6-{luna,sol,terra}-pro`) on the `openai` and `openai-codex` providers: each alias sends the base model id on the wire (`requestModelId`) with the new `reasoningMode: "pro"` marker, and re-derives from the current base rows on every catalog regeneration. +### Changed + +- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header in OpenAI compatible endpoints + +- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header +- Marked direct xAI Grok Chat Completions models for `x-grok-conv-id` prompt-cache affinity. + ## [16.3.14] - 2026-07-09 ### Added diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index d5627ccd5..dcfcd3df9 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -541,7 +541,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.id), emptyLengthFinishIsContextError: provider === "ollama", usesOpenAIToolCallIdLimit: provider === "openai", - promptCacheSessionHeader: undefined, + promptCacheSessionHeader: isGrok ? "x-grok-conv-id" : undefined, dropThinkingWhenReasoningEffort: provider === "fireworks", }; From e2e5d8835071d32bb8654664522b061fb484ab78 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:29:08 +0200 Subject: [PATCH 014/205] feat(coding-agent/prompts): consolidated testing guidance into system prompt - Deleted the standalone Tester subagent file. - Updated the main system prompt to incorporate comprehensive testing requirements and quality standards. - Removed the Tester agent registration from the agent definitions. --- packages/coding-agent/CHANGELOG.md | 6 + .../coding-agent/src/prompts/agents/tester.md | 111 ------------------ .../src/prompts/system/system-prompt.md | 8 +- packages/coding-agent/src/task/agents.ts | 2 - 4 files changed, 11 insertions(+), 116 deletions(-) delete mode 100644 packages/coding-agent/src/prompts/agents/tester.md diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 78d22734a..c1635dcd6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,12 @@ ## [Unreleased] +### Changed + +- Integrated testing guidance directly into the main system prompt for improved workflow cohesion + +- Moved testing guidance into the main system prompt and removed the bundled Tester subagent. + ## [16.3.14] - 2026-07-09 ### Fixed diff --git a/packages/coding-agent/src/prompts/agents/tester.md b/packages/coding-agent/src/prompts/agents/tester.md deleted file mode 100644 index bdd9a96bf..000000000 --- a/packages/coding-agent/src/prompts/agents/tester.md +++ /dev/null @@ -1,111 +0,0 @@ ---- -name: Tester -description: Authoritative test writer. ALWAYS delegate test authoring to this agent — NEVER write tests yourself. Writes high-signal tests defending real contracts (behavior, invariants, edge cases) and refuses worthless tests that assert plumbing or restate the code. -tools: read, grep, glob, bash, edit, write, lsp, ast_grep, ast_edit -spawns: explore -model: pi/task -thinking-level: high ---- - - -RFC 2119 applies to MUST, REQUIRED, SHOULD, RECOMMENDED, MAY, OPTIONAL. `NEVER` and `AVOID` MUST be interpreted as aliases for `MUST NOT` and `SHOULD NOT` respectively. - - -You are a staff test engineer with taste. You write tests that earn their place in the suite and you delete — or refuse to write — tests that don't. You have agency: when asked for coverage that proves nothing, you write the test that would actually catch the bug instead. - - -A test suite is a liability until it pays for itself. Every worthless test is negative value: it costs CI time, blocks honest refactors, and lulls the team into false confidence while the real bug ships. A test's only job is to FAIL when behavior breaks and PASS otherwise. A test that cannot fail for any real defect is noise wearing a green check. You are here because models flood codebases with exactly that noise. You write the opposite. - - - -- The litmus for every test: **name the concrete, externally observable contract it defends** — a behavior, output shape, state transition, error mapping, invariant, or a regression-prone parsing boundary. Cannot name it in one sentence? NEVER write the test. -- Mutation test in your head: if a plausible bug — a flipped condition, an off-by-one, a wrong return value, a dropped case — would still let the test PASS, the test is worthless. Discard it. -- You NEVER write tests that assert plumbing or restate the implementation. The forbidden classes are enumerated in `` and are hard prohibitions. -- You MUST match the repo's existing test conventions — framework, file layout, naming, assertion style. A second convention beside an existing one is PROHIBITED. -- NEVER test defaults (configurations, fallback values, or default environment values). If you are updating/refactoring existing tests that test defaults, you MUST delete those assertions or delete the entire default-testing tests instead. -- You are explicitly ALLOWED to write **no tests at all** if you were spawned for a stupid reason (meaning: the change is trivial—such as docs, comments, types, exports, or simple config; the behavior is already fully covered; or any tests you would write would be worthless, restate plumbing, or test defaults). If so, state this clearly and exit. - - - -NEVER write any of these. Each is a green check that survives real bugs: -- **Config/setter echo.** Setting a value then asserting it reads back (`set(x, 30); expect(get(x)).toBe(30)`) tests the language's assignment, not your code. -- **Source-grep.** Reading an implementation/build file and asserting on its TEXT — `expect(src).toContain("newFn()")`, `.toMatch(/import …/)`, `.not.toContain("oldName")`, "comment says X". Tests how code LOOKS, breaks on rename/reflow, passes while behavior is broken. Enforce structural facts with a type test or lint rule; enforce behavior by running the code. -- **Tautologies.** `expect(true).toBe(true)`, `expect(x).toBe(x)`, asserting a constant equals its literal. -- **Bare no-throw.** `expect(() => f()).not.toThrow()` with no assertion on the result. "It ran" is not a contract. -- **Construction smoke.** "Constructs without error", "package boots", "command starts" — unless that wiring genuinely can't be exercised in-process AND a real failure mode hides there. -- **Mock round-trips.** Asserting a mock was called with the args you just passed it. You tested the mock, not the system. -- **Existence/shape-only.** Non-empty string, length-grew, "field is defined", "returns an object with key Y" — without asserting the VALUE that matters. -- **Default values.** NEVER assert that default configurations, fallback properties, or default environment values match specific literals. A harmless change to a default setting must never break the tests. If you are touching or refactoring existing tests that assert defaults, **delete those assertions or the entire test instead**. -- **Field-wiring.** Asserting an option passed in lands on a property, or that a getter returns the value the constructor stored. Test the downstream BEHAVIOR that depends on it, not the assignment. -- **Duplicate-layer coverage.** Re-proving through mocks what an integration test already proves. Drop the narrower restatement. - -When asked for coverage that would only produce the above, you write the test that actually exercises the behavior, and you state in your result why the requested shape was worthless. - - - -Aim every test at something that can actually break: -- **Behavior & outputs** — given input, the observable result (return value, emitted event, written file, error surfaced). -- **State transitions** — the legal and illegal moves of a stateful component; one test per invariant or transition, not one per field touched. -- **Invariants across fields** — relationships that MUST hold (sorted output stays sorted, sum of parts equals total, encode∘decode is identity). -- **Edge & boundary values** — zero, empty, one, max, negative, off-by-one, overflow, unicode, the value just inside and just outside a limit. -- **Precedence & resolution** — arg beats env beats default; later override wins; first-match-wins. -- **Error paths** — trigger the REAL failure (bad input, missing dep, denied permission) and assert the surfaced contract (error type, message mapping, exit code). NEVER instantiate the error class directly or inspect internal metadata. -- **Regression-prone parsing boundaries** — the exact bytes where a parser/serializer historically broke; pin past regressions with a named case. - - - -Reach for the right shape; do not reinvent what the repo's framework already gives you. -- **Table-driven tests.** One body, many `{ name, input, expected }` rows covering boundaries and equivalence classes plus error cases. Name every row so a failure points at the case. The default shape for any function with a clear input→output mapping. -- **Subtests.** Group related cases under one parent with isolated setup and independent failure reporting. Prefer over many tiny near-duplicate test functions. -- **Property-based tests.** Assert invariants over generated inputs — round-trip identity, idempotence (`f(f(x)) == f(x)`), commutativity, monotonicity, "never panics and output stays well-formed". Catches cases you wouldn't enumerate by hand. -- **Deterministic randomness.** Seed every generator and PRINT the seed on failure so a red run reproduces exactly. NEVER use an unseeded clock-derived source — flaky tests are worse than no tests. -- **Fuzz tests.** For parsers, decoders, deserializers, anything eating untrusted bytes: feed mutated/random input, assert no crash and that invariants hold. Seed the corpus from known-tricky inputs and every past regression. -- **Benchmarks.** ONLY when performance is part of the contract. Measure the operation, not setup; consume the result so it isn't optimized away; compare against a baseline or threshold. A benchmark that asserts nothing is documentation, not a test. -- **Golden/snapshot.** Only for genuinely stable, human-reviewed output where exact bytes are the contract (codegen, serialized formats). NEVER snapshot volatile or incidental output — it becomes a rubber stamp nobody reads. - - - -- **Test through the public API**, the way a real consumer calls it. Place tests in an EXTERNAL test package/module (separate namespace, no access to internals) so the compiler forbids reaching past the contract. This is the default and it forces you to test what callers depend on. -- **Internal (white-box) tests only for private invariants with no observable surface** — e.g. a balancing property of an internal tree, a cache eviction order. Justify each one; if the invariant has an observable effect, test that effect from outside instead. -- NEVER reach into private state to assert what you could observe through the public surface. Coupling tests to internals is what makes refactors painful and tempts people to delete the suite. - - - -- **Prefer real implementations.** If the dependency is cheap and deterministic, use the real thing. -- **Prefer hand-written fakes over mocking frameworks.** A small in-memory implementation of an interface is type-checked, readable, survives refactors, and tests behavior. Mocking frameworks pull you toward asserting call counts and argument sequences — that is plumbing, and it breaks on every harmless internal change. -- **Mock only true external boundaries** — network, wall clock, filesystem, system randomness, third-party services — and even there a fake beats a mock. Inject the boundary; never patch globals. -- NEVER use module-registry mocking that leaks across test files. Spy on the imported object and restore in teardown. - - - -Tests MUST be full-suite safe and order-independent, not merely file-local safe. -- **No timing dependence.** NEVER `sleep`/`setTimeout`-race to "let it settle". Inject a controllable clock and advance it; wait on a condition, signal, or promise, never a wall-clock duration. Real-time waits are the #1 source of flake. -- **No environment pollution.** NEVER leak env vars, temp files, global singletons, `process.env`/`process.platform`/`Bun.*` mutations, or monkeypatches past the test. Use per-test setup with restore in teardown. A test that passes alone but poisons a later file is broken. -- **Deterministic.** No dependence on map/iteration order, filesystem ordering, locale, timezone, or concurrency interleaving unless that ordering IS the contract under test. -- **Hermetic.** No real network or real time. Each test creates and tears down its own fixtures. - - - -1. **Study the code under test.** Read exact signatures, return types, and error paths with `lsp`/`read` — NEVER guess an API. Spawn `explore` for unfamiliar areas. -2. **Study existing tests.** Find the framework, file layout, naming, fake/fixture helpers, and assertion style. You MUST reuse them. `grep`/`glob` for sibling test files. -3. **Enumerate contracts.** List the observable behaviors, invariants, edge cases, and error mappings worth defending. Drop anything that fails the `` litmus. -4. **Pick the shape** per `` — table, property, fuzz, benchmark, or a focused unit/integration test. -5. **Write the tests**, matching repo conventions exactly. Assert semantic content; assert exact bytes ONLY where downstream parses them. -6. **Run them and verify they have teeth.** Execute the suite with the repo's runner; confirm green. Then confirm each test can FAIL: mentally (or by a throwaway mutation) check that a real defect reddens it. A test you never saw fail is unproven. - - - -- You MUST run the tests you wrote with the project's test command and confirm they pass. -- You MUST confirm they are not vacuous: a test that passes against broken code is a defect you authored. When cheap, perturb the implementation to watch the test fail, then revert. -- Run ONLY the tests you added or touched unless asked for the full suite. -- Report each test by the contract it defends — not "added N tests", but "covers ". - - - -- A test exists to FAIL on a real bug. No nameable contract, or no plausible bug would redden it → NEVER write it. -- NEVER assert plumbing, restate the implementation, or grep the source. Test observable behavior through the public surface. -- No timing races, no environment pollution, deterministic and order-independent — full-suite safe. -- NEVER test defaults. If updating tests that do, delete them instead. -- You are explicitly ALLOWED to write **no tests at all** if you were spawned for a stupid reason (trivial changes, already covered, or if any possible test would be worthless/test defaults). -- You MUST keep going until the tests are written, passing, and proven to have teeth (unless skipped per above). - diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index e90d57c5e..562f7856b 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -182,9 +182,11 @@ EXECUTION WORKFLOW {{#has tools "ask"}}- Ask before destructive commands or deleting code you didn't write.{{else}}- Don't run destructive git commands or delete code you didn't write.{{/has}} # 5. Verify -- NEVER yield non-trivial work without proof: tests, E2E, browsing, or QA. Run only tests you added or modified unless asked otherwise. -- Test behavior, using tester agent where available. Assert logical behavior, not current state. -- Aim at conditional branches, edge values, invariants across fields, and error handling versus silent broken results. +- NEVER yield non-trivial work without proof: tests, E2E, browsing, or QA. +- Every test MUST defend an observable contract and fail on a plausible bug. +- Test behavior, boundaries, invariants, transitions, precedence, and real errors—not plumbing, source text, or incidental defaults. +- Match existing conventions; keep tests deterministic, isolated, and full-suite safe. +- Run only touched tests; small/no-test changes still REQUIRE a focused behavioral smoke test. # 6. Cleanup Changelog, tests, docs, and removing scaffolding are the LAST phase—NEVER skipped, but gated on the request demonstrably working. diff --git a/packages/coding-agent/src/task/agents.ts b/packages/coding-agent/src/task/agents.ts index b9f3bd5ed..d9c72f226 100644 --- a/packages/coding-agent/src/task/agents.ts +++ b/packages/coding-agent/src/task/agents.ts @@ -15,7 +15,6 @@ import librarianMd from "../prompts/agents/librarian.md" with { type: "text" }; import planMd from "../prompts/agents/plan.md" with { type: "text" }; import reviewerMd from "../prompts/agents/reviewer.md" with { type: "text" }; import taskMd from "../prompts/agents/task.md" with { type: "text" }; -import testerMd from "../prompts/agents/tester.md" with { type: "text" }; import type { AgentDefinition, AgentSource } from "./types"; @@ -47,7 +46,6 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ { fileName: "designer.md", template: designerMd }, { fileName: "reviewer.md", template: reviewerMd }, { fileName: "librarian.md", template: librarianMd }, - { fileName: "tester.md", template: testerMd }, { fileName: "task.md", frontmatter: { From 66cd994cbaae392f1fdfe4a29b9b0a4de76d43ba Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:37:09 +0200 Subject: [PATCH 015/205] fix(ai): standardized tier classification and model routing logic - Implemented plan-based tier classification for OpenAI Codex usage to ensure correct account routing. - Updated authentication logic to normalize metadata and prioritize eligible accounts for GPT-5.6 models. - Added fallback mechanisms to ensure standard usage ranking persists when specific tier requirements are not met. - Validated routing behavior and model-specific selection through comprehensive unit test coverage. --- packages/ai/CHANGELOG.md | 5 +- packages/ai/src/auth-storage.ts | 144 ++++++++++++----- .../test/auth-storage-codex-selection.test.ts | 151 ++++++++++++++++++ 3 files changed, 259 insertions(+), 41 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 96c28897b..3812e1ccb 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -9,7 +9,6 @@ ### Added - Added automatic prompt-cache affinity header injection for OpenAI-family chat completions - - Added support for explicit prompt-cache affinity headers in OpenAI-family chat completions - Added OpenAI pro reasoning mode support: models carrying the catalog `reasoningMode: "pro"` marker (GPT-5.6 Pro aliases) send `reasoning: { mode: "pro" }` on OpenAI Responses and Codex Responses requests, alongside the configured effort. The Codex request body now honors `requestModelId` so catalog aliases request the base upstream model id. @@ -19,6 +18,10 @@ ### Fixed +- Improved account routing for GPT-5.6 models to better respect paid tier requirements +- Refined account selection logic to correctly identify plan types from account metadata + +- Fixed OpenAI Codex multi-account routing for GPT-5.6: Sol and Luna requests now prefer Plus-or-higher accounts while Terra remains available to Free/Go accounts; local pro-mode aliases inherit their base model's Codex plan eligibility. - Fixed xAI Grok OAuth login to use xAI's device authorization flow: `/login` now opens the verification URL, displays the device code, and polls for approval instead of asking for a pasted redirect or linking to Hermes Agent documentation. ## [16.3.14] - 2026-07-09 diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 315d43c48..2c094515b 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -761,25 +761,82 @@ function isAbortSignalOption( return typeof value === "object" && value !== null && "aborted" in value && "addEventListener" in value; } -function requiresOpenAICodexProModel(provider: string, modelId: string | undefined): boolean { - return provider === "openai-codex" && typeof modelId === "string" && modelId.includes("-spark"); +type OpenAICodexPlanRequirement = "none" | "paid" | "pro"; +type OpenAICodexPlanClass = "free" | "paid" | "pro" | "unknown"; + +const GPT_56_PAID_CODEX_MODEL_PATTERN = /^gpt-5\.6-(?:sol|luna)(?:-pro)?$/; +const OPENAI_CODEX_PRO_PLAN_TOKENS: Record = { + pro: true, +}; +const OPENAI_CODEX_PAID_PLAN_TOKENS: Record = { + plus: true, + business: true, + team: true, + enterprise: true, + edu: true, + education: true, + teacher: true, + teachers: true, + health: true, + gov: true, + government: true, +}; +const OPENAI_CODEX_FREE_PLAN_TOKENS: Record = { + free: true, + go: true, +}; + +/** + * Account tier needed for model-aware Codex OAuth routing. + * + * GPT-5.6 Terra (including its local pro-mode alias) remains available on every + * plan. Sol and Luna pro-mode aliases inherit their base models' paid tier; + * only Spark currently has a documented Pro-plan preference in Codex. + */ +function resolveOpenAICodexPlanRequirement(provider: string, modelId: string | undefined): OpenAICodexPlanRequirement { + if (provider !== "openai-codex" || typeof modelId !== "string") return "none"; + const separator = modelId.lastIndexOf("/"); + const bareModelId = (separator === -1 ? modelId : modelId.slice(separator + 1)).toLowerCase(); + if (bareModelId.includes("-spark")) return "pro"; + if (bareModelId === "gpt-5.6" || GPT_56_PAID_CODEX_MODEL_PATTERN.test(bareModelId)) return "paid"; + return "none"; } function getUsagePlanType(report: UsageReport | null): string | undefined { const metadata = report?.metadata; - if (!metadata || typeof metadata !== "object" || Array.isArray(metadata)) return undefined; - const planType = (metadata as { planType?: unknown }).planType; - return typeof planType === "string" ? planType.toLowerCase() : undefined; + if (!metadata) return undefined; + const planType = metadata.planType; + if (typeof planType !== "string") return undefined; + const normalized = planType + .trim() + .toLowerCase() + .replace(/[\s-]+/g, "_"); + return normalized.startsWith("chatgpt_") ? normalized.slice("chatgpt_".length) : normalized; } -function getOpenAICodexPlanPriority(report: UsageReport | null): number { +function classifyOpenAICodexPlan(report: UsageReport | null): OpenAICodexPlanClass { const planType = getUsagePlanType(report); - if (!planType) return 1; - return planType.includes("pro") ? 0 : 2; + if (!planType) return "unknown"; + const tokens = planType.split("_"); + if (tokens.some(token => OPENAI_CODEX_PRO_PLAN_TOKENS[token] === true)) return "pro"; + if (tokens.some(token => OPENAI_CODEX_PAID_PLAN_TOKENS[token] === true)) return "paid"; + if (tokens.some(token => OPENAI_CODEX_FREE_PLAN_TOKENS[token] === true)) return "free"; + return "unknown"; } -function hasOpenAICodexProPlan(report: UsageReport | null): boolean { - return getUsagePlanType(report)?.includes("pro") === true; +function getOpenAICodexPlanEligibility( + report: UsageReport | null, + requirement: OpenAICodexPlanRequirement, +): boolean | undefined { + if (requirement === "none") return true; + const planClass = classifyOpenAICodexPlan(report); + if (planClass === "unknown") return undefined; + return requirement === "paid" ? planClass !== "free" : planClass === "pro"; +} + +function getOpenAICodexPlanPriority(report: UsageReport | null, requirement: OpenAICodexPlanRequirement): number { + const eligibility = getOpenAICodexPlanEligibility(report, requirement); + return eligibility === true ? 0 : eligibility === undefined ? 1 : 2; } function compareUsageRankingMetric(left: number, right: number): number { @@ -3186,8 +3243,7 @@ export class AuthStorage { #compareRankedOAuthCandidatePriority( left: RankedOAuthCandidate, right: RankedOAuthCandidate, - provider: string, - modelId: string | undefined, + planRequirement: OpenAICodexPlanRequirement, ): number { if (left.blocked !== right.blocked) return left.blocked ? 1 : -1; if (left.blocked && right.blocked) { @@ -3196,7 +3252,7 @@ export class AuthStorage { if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil; return 0; } - if (requiresOpenAICodexProModel(provider, modelId) && left.planPriority !== right.planPriority) { + if (planRequirement !== "none" && left.planPriority !== right.planPriority) { return left.planPriority - right.planPriority; } if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; @@ -3214,20 +3270,18 @@ export class AuthStorage { #compareRankedOAuthCandidates( left: RankedOAuthCandidate, right: RankedOAuthCandidate, - provider: string, - modelId: string | undefined, + planRequirement: OpenAICodexPlanRequirement, ): number { - const priority = this.#compareRankedOAuthCandidatePriority(left, right, provider, modelId); + const priority = this.#compareRankedOAuthCandidatePriority(left, right, planRequirement); return priority !== 0 ? priority : left.orderPos - right.orderPos; } #orderRankedOAuthCandidates( candidates: RankedOAuthCandidate[], sessionId: string | undefined, - provider: string, - modelId: string | undefined, + planRequirement: OpenAICodexPlanRequirement, ): OAuthCandidate[] { - candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, provider, modelId)); + candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, planRequirement)); if (!sessionId) { return candidates.map(candidate => ({ selection: candidate.selection, @@ -3252,7 +3306,7 @@ export class AuthStorage { for (const candidate of unblocked) { if ( candidate !== previous && - this.#compareRankedOAuthCandidatePriority(previous, candidate, provider, modelId) !== 0 + this.#compareRankedOAuthCandidatePriority(previous, candidate, planRequirement) !== 0 ) { bucketIndex += 1; } @@ -3297,6 +3351,7 @@ export class AuthStorage { providerKey: string; provider: string; order: number[]; + planRequirement: OpenAICodexPlanRequirement; credentials: OAuthSelection[]; options?: AuthApiKeyOptions; sessionId?: string; @@ -3378,7 +3433,7 @@ export class AuthStorage { blocked, blockedUntil, hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false, - planPriority: getOpenAICodexPlanPriority(usage), + planPriority: getOpenAICodexPlanPriority(usage, args.planRequirement), secondaryUsed: this.#normalizeUsageFraction(secondaryTarget), secondaryDrainRate: this.#computeWindowDrainRate( secondaryTarget, @@ -3390,7 +3445,7 @@ export class AuthStorage { orderPos, }); } - return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.provider, args.options?.modelId); + return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.planRequirement); } /** @@ -3418,8 +3473,9 @@ export class AuthStorage { const strategy = this.#rankingStrategyResolver?.(provider); const rankingContext: CredentialRankingContext = { modelId: options?.modelId }; const blockScope = strategy?.blockScope?.(rankingContext); - const requiresProModel = requiresOpenAICodexProModel(provider, options?.modelId); - const checkUsage = strategy !== undefined && (credentials.length > 1 || requiresProModel); + const planRequirement = resolveOpenAICodexPlanRequirement(provider, options?.modelId); + const hasPlanRequirement = planRequirement !== "none"; + const checkUsage = strategy !== undefined && (credentials.length > 1 || hasPlanRequirement); const sessionCredential = this.#getSessionCredential(provider, sessionId); const sessionPreferredIndex = sessionCredential?.type === "oauth" ? sessionCredential.index : undefined; const sessionPreferredCredential = @@ -3438,12 +3494,13 @@ export class AuthStorage { sessionPreferredIndex !== undefined && sessionPreferredCanRefreshOrUse && !this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope); - const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel); + const shouldRank = checkUsage && (!sessionPreferredIsAvailable || hasPlanRequirement); const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; const candidates = shouldRank ? await this.#rankOAuthSelections({ providerKey, provider, + planRequirement, order: rankingOrder, credentials, options, @@ -3457,7 +3514,7 @@ export class AuthStorage { .filter((selection): selection is { credential: OAuthCredential; index: number } => Boolean(selection)) .map(selection => ({ selection, usage: null, usageChecked: false })); - if (sessionPreferredIndex !== undefined && !requiresProModel) { + if (sessionPreferredIndex !== undefined && !hasPlanRequirement) { const sessionPreferredCandidate = candidates.findIndex( candidate => !this.#isCredentialBlocked(provider, providerKey, candidate.selection.index, blockScope) && @@ -3537,10 +3594,12 @@ export class AuthStorage { }), ); - // Skip the Pro-plan filter when no candidate is confirmed Pro, so users with only - // non-Pro accounts can still attempt Spark requests (e.g. trial/grandfathered access). - const enforceProRequirement = - requiresProModel && candidates.some(candidate => hasOpenAICodexProPlan(candidate.usage)); + // Enforce a tier only when at least one account is confirmed eligible. If + // every report is unknown or ineligible, preserve trial/grandfathered access + // by allowing the normal candidate fallback to attempt the request. + const enforcePlanRequirement = + hasPlanRequirement && + candidates.some(candidate => getOpenAICodexPlanEligibility(candidate.usage, planRequirement) === true); const fallback = candidates[0]; @@ -3556,7 +3615,8 @@ export class AuthStorage { allowBlocked: false, prefetchedUsage: candidate.usage, usagePrechecked: candidate.usageChecked, - enforceProRequirement, + planRequirement, + enforcePlanRequirement, strategy, rankingContext, blockScope, @@ -3571,7 +3631,8 @@ export class AuthStorage { allowBlocked: true, prefetchedUsage: fallback.usage, usagePrechecked: fallback.usageChecked, - enforceProRequirement, + planRequirement, + enforcePlanRequirement, strategy, rankingContext, blockScope, @@ -3709,7 +3770,8 @@ export class AuthStorage { allowBlocked: boolean; prefetchedUsage?: UsageReport | null; usagePrechecked?: boolean; - enforceProRequirement?: boolean; + planRequirement?: OpenAICodexPlanRequirement; + enforcePlanRequirement?: boolean; strategy?: CredentialRankingStrategy; rankingContext?: CredentialRankingContext; blockScope?: string; @@ -3722,7 +3784,8 @@ export class AuthStorage { allowBlocked, prefetchedUsage = null, usagePrechecked = false, - enforceProRequirement, + planRequirement: providedPlanRequirement, + enforcePlanRequirement, strategy, rankingContext, blockScope, @@ -3741,12 +3804,13 @@ export class AuthStorage { // refresh / persist / CAS-disable addresses the row by this stable id. const credentialId = this.#getStoredCredentials(provider)[selection.index]?.id; - const requiresProModel = requiresOpenAICodexProModel(provider, options?.modelId); - const applyProFilter = enforceProRequirement ?? requiresProModel; + const planRequirement = providedPlanRequirement ?? resolveOpenAICodexPlanRequirement(provider, options?.modelId); + const hasPlanRequirement = planRequirement !== "none"; + const applyPlanFilter = enforcePlanRequirement ?? hasPlanRequirement; let usage: UsageReport | null = null; let usageChecked = false; - if ((checkUsage && !allowBlocked) || requiresProModel) { + if ((checkUsage && !allowBlocked) || hasPlanRequirement) { if (usagePrechecked) { usage = prefetchedUsage; usageChecked = true; @@ -3757,7 +3821,7 @@ export class AuthStorage { }); usageChecked = true; } - if (applyProFilter && !hasOpenAICodexProPlan(usage)) { + if (applyPlanFilter && getOpenAICodexPlanEligibility(usage, planRequirement) !== true) { return undefined; } if (checkUsage && !allowBlocked && usage && strategy && rankingContext) { @@ -3825,7 +3889,7 @@ export class AuthStorage { } else { this.#replaceCredentialAt(provider, selection.index, updated); } - if ((checkUsage && !allowBlocked) || requiresProModel) { + if ((checkUsage && !allowBlocked) || hasPlanRequirement) { const sameAccount = selection.credential.accountId === updated.accountId; if (!usageChecked || !sameAccount) { usage = await this.#getUsageReport(provider, updated, { @@ -3834,7 +3898,7 @@ export class AuthStorage { }); usageChecked = true; } - if (applyProFilter && !hasOpenAICodexProPlan(usage)) { + if (applyPlanFilter && getOpenAICodexPlanEligibility(usage, planRequirement) !== true) { return undefined; } if (checkUsage && !allowBlocked && usage && strategy && rankingContext) { diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6bf0ff584..a0ff1e114 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -724,6 +724,157 @@ describe("AuthStorage codex oauth ranking", () => { expect(apiKey).toBe("api-acct-solo"); }); + test.each([ + ["gpt-5.6-sol", "free", "plus"], + ["gpt-5.6-luna", "go", "business"], + ["gpt-5.6-sol-pro", "free", "team"], + ])("%s routes away from a less-used %s account to an eligible %s account", async (modelId, freePlan, paidPlan) => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-free", "free@example.com") }, + { type: "oauth", ...createCredential("acct-paid", "paid@example.com") }, + ]); + + usageByAccount.set( + "acct-free", + createCodexUsageReport({ + accountId: "acct-free", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: freePlan, email: "free@example.com" }, + }), + ); + usageByAccount.set( + "acct-paid", + createCodexUsageReport({ + accountId: "acct-paid", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: paidPlan, email: "paid@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId }); + expect(apiKey).toBe("api-acct-paid"); + }); + + test.each([ + ["gpt-5.6-terra", "free", "enterprise"], + ["gpt-5.6-terra-pro", "go", "pro"], + ])("%s keeps a less-used %s account in ordinary ranking ahead of %s", async (modelId, lowUsagePlan, highUsagePlan) => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-low-usage", "low-usage@example.com") }, + { type: "oauth", ...createCredential("acct-high-usage", "high-usage@example.com") }, + ]); + + usageByAccount.set( + "acct-low-usage", + createCodexUsageReport({ + accountId: "acct-low-usage", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: lowUsagePlan, email: "low-usage@example.com" }, + }), + ); + usageByAccount.set( + "acct-high-usage", + createCodexUsageReport({ + accountId: "acct-high-usage", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: highUsagePlan, email: "high-usage@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId }); + expect(apiKey).toBe("api-acct-low-usage"); + }); + + test("reranks a Terra session on a Go account when it switches to Sol", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-go", "go@example.com") }, + { type: "oauth", ...createCredential("acct-business", "business@example.com") }, + ]); + + usageByAccount.set( + "acct-go", + createCodexUsageReport({ + accountId: "acct-go", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "go", email: "go@example.com" }, + }), + ); + usageByAccount.set( + "acct-business", + createCodexUsageReport({ + accountId: "acct-business", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "business", email: "business@example.com" }, + }), + ); + + let terraSession: string | undefined; + let terraApiKey: string | undefined; + for (let index = 0; index < 100; index += 1) { + const sessionId = `session-terra-to-sol-${index}`; + const apiKey = await authStorage.getApiKey("openai-codex", sessionId, { + modelId: "gpt-5.6-terra", + }); + if (apiKey === "api-acct-go") { + terraSession = sessionId; + terraApiKey = apiKey; + break; + } + } + expect(terraApiKey).toBe("api-acct-go"); + if (!terraSession) throw new Error("expected Terra to select the lower-usage Go account"); + + const solApiKey = await authStorage.getApiKey("openai-codex", terraSession, { + modelId: "gpt-5.6-sol", + }); + expect(solApiKey).toBe("api-acct-business"); + }); + + test("falls back by ordinary usage ranking for Sol when no account is confirmed paid", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-free", "free@example.com") }, + { type: "oauth", ...createCredential("acct-go", "go@example.com") }, + ]); + + usageByAccount.set( + "acct-free", + createCodexUsageReport({ + accountId: "acct-free", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "free", email: "free@example.com" }, + }), + ); + usageByAccount.set( + "acct-go", + createCodexUsageReport({ + accountId: "acct-go", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "go", email: "go@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { + modelId: "gpt-5.6-sol", + }); + expect(apiKey).toBe("api-acct-go"); + }); + test("prefers Pro accounts for codex spark models over Plus accounts", async () => { if (!authStorage) throw new Error("test setup failed"); From 46b8ee737f5f5165321e5f856a4e1b31d56e1758 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:39:18 +0200 Subject: [PATCH 016/205] feat(catalog): added Grok 4.5 model support - Added Grok 4.5 to the model catalog and identity helper. - Updated pricing and configuration settings for existing Grok models. --- packages/catalog/CHANGELOG.md | 6 +- packages/catalog/src/identity/family.ts | 2 +- packages/catalog/src/models.json | 34 +++++- .../src/provider-models/openai-compat.ts | 1 + packages/catalog/test/identity-family.test.ts | 1 + .../natives/test/issue-4866-repro.test.ts | 109 ------------------ 6 files changed, 39 insertions(+), 114 deletions(-) delete mode 100644 packages/natives/test/issue-4866-repro.test.ts diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 144a11730..cde47ac31 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -4,6 +4,8 @@ ### Added +- Added support for Grok 4.5 model + - Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants - Added `meta/muse-spark-1.1` model support - Added support for thinking modes on `poolside/laguna` models @@ -11,8 +13,10 @@ ### Changed -- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header in OpenAI compatible endpoints +- Updated cache read costs for Grok models +- Reduced max token limit for Grok 4.3 model +- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header in OpenAI compatible endpoints - Enabled prompt cache affinity for Grok models via the x-grok-conv-id header - Marked direct xAI Grok Chat Completions models for `x-grok-conv-id` prompt-cache affinity. diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index b62bfe433..aa206cd9f 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -78,7 +78,7 @@ export const isMimoModelIdOrName = memo((value: string): boolean => { return value.toLowerCase().includes("mimo"); }); -const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const; +const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3", "grok-4.5"] as const; /** * Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 67ed8c8e0..0d7cc7d3d 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -64651,7 +64651,7 @@ "cost": { "input": 0.65, "output": 3.41, - "cacheRead": 0.14, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, @@ -67788,7 +67788,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 512000, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -68452,7 +68452,7 @@ "cost": { "input": 0.65, "output": 3.41, - "cacheRead": 0.14, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, @@ -85686,6 +85686,34 @@ "omitReasoningEffort": false } }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-responses", + "provider": "xai-oauth", + "baseUrl": "https://api.x.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": true + } + }, "grok-build": { "id": "grok-build", "name": "Grok Build", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 1877fa7d2..b3e4af403 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1039,6 +1039,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [ input: ["text", "image"], }, { id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] }, + { id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] }, // grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default. { id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" }, { diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts index 716731082..c0748fd8f 100644 --- a/packages/catalog/test/identity-family.test.ts +++ b/packages/catalog/test/identity-family.test.ts @@ -265,6 +265,7 @@ describe("isGrokReasoningEffortCapable", () => { expect(isGrokReasoningEffortCapable("grok-3-mini")).toBe(true); expect(isGrokReasoningEffortCapable("grok-4.20-multi-agent")).toBe(true); expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.3")).toBe(true); + expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.5")).toBe(true); expect(isGrokReasoningEffortCapable("openrouter/xai/grok-3-mini")).toBe(true); }); diff --git a/packages/natives/test/issue-4866-repro.test.ts b/packages/natives/test/issue-4866-repro.test.ts deleted file mode 100644 index b26e97fdb..000000000 --- a/packages/natives/test/issue-4866-repro.test.ts +++ /dev/null @@ -1,109 +0,0 @@ -/** - * Regression for https://github.com/can1357/oh-my-pi/issues/4866. - * - * "When bash command times out, it exits/crashes OMP as a whole" (WSL). - * - * Root cause: the native shell output bridge (`bridge_chunks` in - * `crates/pi-natives/src/shell.rs` + `emit_chunk` in - * `crates/pi-shell/src/shell.rs`) queued decoded output chunks into an - * unbounded cross-thread channel and fired the JS threadsafe function - * non-blocking, with no backpressure. A producer outrunning the JS consumer - * (`yes | cat` runs as in-process uutils builtins at memory speed; any - * output-heavy long task qualifies) ballooned the native queue by gigabytes - * before the timeout fired, and the callback flood then kept the JS event - * loop saturated so the deadline machinery ran tens of seconds late. - * Measured on the pre-fix baseline (macOS arm64): a `timeoutMs: 1500` run - * through the bash executor resolved after ~30-36 s having forwarded ~6.9 GB, - * with process RSS pinned at ~7 GB. On WSL's memory-capped VM that backlog - * trips the Linux OOM killer, which SIGKILLs the whole OMP process — the - * reported "crashes OMP as a whole". - * - * This test models the real consumer (OutputSink sanitize/tail/render work) - * with a deliberately slow `onChunk` (~1 ms per callback) and pins the fixed - * contract for both the one-shot (`executeShell`) and persistent-session - * (`Shell.run`) paths: - * 1. The run resolves near its deadline (raced against a generous window) - * instead of being dragged out by an unbounded backlog drain. On the - * pre-fix bridge the drain alone needs minutes (tens of thousands of - * queued 64 KiB batches through a ~1 ms consumer). - * 2. Native memory stays bounded: RSS growth over the run stays far under - * the gigabytes the unbounded queue accumulated (bounded(64) queue × - * 64 KiB batches plus JS churn). - * 3. The run still reports `timedOut`, so timeout annotation and session - * quarantine behave as before. - * - * Bounds carry >4x headroom on both sides of every threshold (fixed path - * measured: resolve ≈1 s, RSS delta ≈60 MiB; baseline: unresolved at 6 s, - * RSS delta ≥2 GiB), so the test stays robust on slow CI hosts while the - * failure mode overshoots by orders of magnitude. - */ -import { describe, expect, it } from "bun:test"; -import { executeShell, Shell, type ShellRunResult } from "../native/index.js"; - -/** `yes` and `cat` are in-process uutils builtins: output is produced at - * memory speed, which is what made the unbounded bridge lethal. */ -const FAST_PRODUCER = "yes issue-4866-crash-line | cat"; -const TIMEOUT_MS = 800; -/** Window the timed-out run must resolve within (fixed path: ~1 s; pre-fix - * baseline is still draining its multi-GB backlog minutes later). */ -const RESOLVE_WINDOW_MS = 8_000; -/** RSS growth budget. Fixed path: tens of MiB. Pre-fix: multiple GiB. */ -const MAX_RSS_DELTA_BYTES = 512 * 1024 * 1024; -/** Per-callback consumer cost emulating OutputSink/TUI work. */ -const CONSUMER_STALL_MS = 1; -const TEST_BUDGET_MS = 60_000; - -const posixIt = process.platform === "win32" ? it.skip : it; - -// Real-clock integration test (ts-no-test-timers exception): the run under -// test is a native tokio shell execution behind the N-API boundary — fake JS -// timers cannot advance the native runtime's clock, and the defect being -// pinned is precisely a real-time liveness failure (the JS event loop and -// deadline machinery starved by the callback flood). The stall emulates -// synchronous per-callback consumer cost (CPU work, not scheduling), and the -// resolve window is a liveness bound, not a synchronization guess. -async function runTimedOutFastProducer( - run: (onChunk: (err: Error | null, chunk: string) => void) => Promise, -): Promise { - const rssBefore = process.memoryUsage.rss(); - const slowConsumer = (_err: Error | null, chunk: string) => { - if (chunk) Bun.sleepSync(CONSUMER_STALL_MS); - }; - - const settled = run(slowConsumer).then(result => ({ done: true as const, result })); - const raced = await Promise.race([settled, Bun.sleep(RESOLVE_WINDOW_MS).then(() => ({ done: false as const }))]); - const rssDelta = process.memoryUsage.rss() - rssBefore; - - // (2) Bounded native memory — the unbounded bridge queued gigabytes here. - expect(rssDelta).toBeLessThan(MAX_RSS_DELTA_BYTES); - // (1) Timely resolution — the unbounded bridge dragged the run out for - // minutes past its deadline. - expect(raced.done).toBe(true); - if (raced.done) { - // (3) Timeout is still reported as such. - expect(raced.result.timedOut).toBe(true); - } -} - -describe("issue 4866: bash timeout must not flood the output bridge", () => { - posixIt( - "one-shot executeShell: fast producer with slow consumer times out near its deadline with bounded memory", - async () => { - await runTimedOutFastProducer(onChunk => - executeShell({ command: FAST_PRODUCER, timeoutMs: TIMEOUT_MS }, onChunk), - ); - }, - TEST_BUDGET_MS, - ); - - posixIt( - "persistent Shell.run: fast producer with slow consumer times out near its deadline with bounded memory", - async () => { - const shell = new Shell(); - await runTimedOutFastProducer(onChunk => - shell.run({ command: FAST_PRODUCER, timeoutMs: TIMEOUT_MS }, onChunk), - ); - }, - TEST_BUDGET_MS, - ); -}); From e8d0a93db61e756361aad84d4bcbc2dd2db88973 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:44:35 +0200 Subject: [PATCH 017/205] chore: bump version to 16.3.15 --- Cargo.lock | 10 +++--- Cargo.toml | 2 +- bun.lock | 52 +++++++++++++-------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 24 ++++++------- packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 3 +- packages/ai/package.json | 2 +- packages/catalog/CHANGELOG.md | 4 +-- packages/catalog/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 3 +- packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 23 files changed, 66 insertions(+), 64 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index a115a7785..125f5b658 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2881,7 +2881,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.3.14" +version = "16.3.15" dependencies = [ "anyhow", "ast-grep-core", @@ -2950,7 +2950,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.3.14" +version = "16.3.15" dependencies = [ "async-trait", "libc", @@ -2962,7 +2962,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.3.14" +version = "16.3.15" dependencies = [ "anyhow", "arboard", @@ -3015,7 +3015,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.3.14" +version = "16.3.15" dependencies = [ "anyhow", "brush-builtins", @@ -3064,7 +3064,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.3.14" +version = "16.3.15" dependencies = [ "dashmap", "globset", diff --git a/Cargo.toml b/Cargo.toml index 4689b4be4..652fd749f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.3.14" +version = "16.3.15" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index 0d17797d5..d179b61c4 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.14", + "version": "16.3.15", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.14", + "version": "16.3.15", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.3.14", + "version": "16.3.15", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.3.14", + "version": "16.3.15", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.3.14", + "version": "16.3.15", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.3.14", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.3.14", + "version": "16.3.15", "devDependencies": { "@types/bun": "catalog:", }, @@ -338,18 +338,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.14", - "@oh-my-pi/omp-stats": "16.3.14", - "@oh-my-pi/pi-agent-core": "16.3.14", - "@oh-my-pi/pi-ai": "16.3.14", - "@oh-my-pi/pi-catalog": "16.3.14", - "@oh-my-pi/pi-coding-agent": "16.3.14", - "@oh-my-pi/pi-mnemopi": "16.3.14", - "@oh-my-pi/pi-natives": "16.3.14", - "@oh-my-pi/pi-tui": "16.3.14", - "@oh-my-pi/pi-utils": "16.3.14", - "@oh-my-pi/pi-wire": "16.3.14", - "@oh-my-pi/snapcompact": "16.3.14", + "@oh-my-pi/hashline": "16.3.15", + "@oh-my-pi/omp-stats": "16.3.15", + "@oh-my-pi/pi-agent-core": "16.3.15", + "@oh-my-pi/pi-ai": "16.3.15", + "@oh-my-pi/pi-catalog": "16.3.15", + "@oh-my-pi/pi-coding-agent": "16.3.15", + "@oh-my-pi/pi-mnemopi": "16.3.15", + "@oh-my-pi/pi-natives": "16.3.15", + "@oh-my-pi/pi-tui": "16.3.15", + "@oh-my-pi/pi-utils": "16.3.15", + "@oh-my-pi/pi-wire": "16.3.15", + "@oh-my-pi/snapcompact": "16.3.15", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -791,7 +791,7 @@ "@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.9.0", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.9.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/sdk-trace-base": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-ec9a7ps37huy5itYk0MalaZdSLlM6AXWp/FhtEjgMpp5leEGojBDvAl/UWttQnkMZOvFHKzRESn8TD3yKTF5nQ=="], - "@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.41.1", "", {}, "sha512-/UhIkaZgPutTFmQ7RnIJGgDXZmtEJ7Dvi86xNTFWcnRxVRNk/aotsqDJYeEvDP+FSMB2SdW+pQzNMcWP0rwuNA=="], + "@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.42.0", "", {}, "sha512-icc5xCzndZfhuJMy5oqk5AvloWquR7jtae74qzpkKkhGp8BivK+oCcEXgGnjCdTfp8hA44l+w8gE8yYJbocJJw=="], "@oxc-project/types": ["@oxc-project/types@0.138.0", "", {}, "sha512-1a7ZKmrRTCoN1XMZ4L0PyyqrMnrNlLyPuOkdSX2MZg7IiIGRUyurNhAm73ptDOraoBcIordsIGKNPKUzy3ZmfA=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 3a0964bf5..5b93f4516 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_3_14")] +#[napi(js_name = "__piNativesV16_3_15")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index 415f8d5a6..a63330c5c 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.14", - "@oh-my-pi/omp-stats": "16.3.14", - "@oh-my-pi/pi-agent-core": "16.3.14", - "@oh-my-pi/pi-ai": "16.3.14", - "@oh-my-pi/pi-catalog": "16.3.14", - "@oh-my-pi/pi-coding-agent": "16.3.14", - "@oh-my-pi/pi-mnemopi": "16.3.14", - "@oh-my-pi/pi-natives": "16.3.14", - "@oh-my-pi/pi-tui": "16.3.14", - "@oh-my-pi/pi-utils": "16.3.14", - "@oh-my-pi/pi-wire": "16.3.14", - "@oh-my-pi/snapcompact": "16.3.14", + "@oh-my-pi/hashline": "16.3.15", + "@oh-my-pi/omp-stats": "16.3.15", + "@oh-my-pi/pi-agent-core": "16.3.15", + "@oh-my-pi/pi-ai": "16.3.15", + "@oh-my-pi/pi-catalog": "16.3.15", + "@oh-my-pi/pi-coding-agent": "16.3.15", + "@oh-my-pi/pi-mnemopi": "16.3.15", + "@oh-my-pi/pi-natives": "16.3.15", + "@oh-my-pi/pi-tui": "16.3.15", + "@oh-my-pi/pi-utils": "16.3.15", + "@oh-my-pi/pi-wire": "16.3.15", + "@oh-my-pi/snapcompact": "16.3.15", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index fd21f9d17..0e5b1961c 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.14", + "version": "16.3.15", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3812e1ccb..03dd313da 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.3.15] - 2026-07-09 + ### Breaking Changes - Renamed `OpenAIResponsesCacheOptions`, `normalizeOpenAIResponsesPromptCacheKey`, and `getOpenAIResponsesPromptCacheKey` to the endpoint-neutral `OpenAICacheOptions`, `normalizeOpenAIPromptCacheKey`, and `getOpenAIPromptCacheKey`. @@ -20,7 +22,6 @@ - Improved account routing for GPT-5.6 models to better respect paid tier requirements - Refined account selection logic to correctly identify plan types from account metadata - - Fixed OpenAI Codex multi-account routing for GPT-5.6: Sol and Luna requests now prefer Plus-or-higher accounts while Terra remains available to Free/Go accounts; local pro-mode aliases inherit their base model's Codex plan eligibility. - Fixed xAI Grok OAuth login to use xAI's device authorization flow: `/login` now opens the verification URL, displays the device code, and polls for approval instead of asking for a pasted redirect or linking to Hermes Agent documentation. diff --git a/packages/ai/package.json b/packages/ai/package.json index 438c2f1f1..1124f3f13 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.3.14", + "version": "16.3.15", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index cde47ac31..b733a706c 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,10 +2,11 @@ ## [Unreleased] +## [16.3.15] - 2026-07-09 + ### Added - Added support for Grok 4.5 model - - Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants - Added `meta/muse-spark-1.1` model support - Added support for thinking modes on `poolside/laguna` models @@ -15,7 +16,6 @@ - Updated cache read costs for Grok models - Reduced max token limit for Grok 4.3 model - - Enabled prompt cache affinity for Grok models via the x-grok-conv-id header in OpenAI compatible endpoints - Enabled prompt cache affinity for Grok models via the x-grok-conv-id header - Marked direct xAI Grok Chat Completions models for `x-grok-conv-id` prompt-cache affinity. diff --git a/packages/catalog/package.json b/packages/catalog/package.json index dc1388a19..653aac2e6 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.3.14", + "version": "16.3.15", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c1635dcd6..6c5d526e2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,10 +2,11 @@ ## [Unreleased] +## [16.3.15] - 2026-07-09 + ### Changed - Integrated testing guidance directly into the main system prompt for improved workflow cohesion - - Moved testing guidance into the main system prompt and removed the bundled Tester subagent. ## [16.3.14] - 2026-07-09 diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 9094d0894..c66b13e2f 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.14", + "version": "16.3.15", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index a3c5f7cf6..73c31c71a 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.3.14", + "version": "16.3.15", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 9a4845089..b5a62f06e 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.14", + "version": "16.3.15", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 3220cfb8c..bb54f9bf5 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -170,7 +170,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_3_14(): void +export declare function __piNativesV16_3_15(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 09c13c442..f439ff2c7 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_3_14 = nativeBindings.__piNativesV16_3_14; +export const __piNativesV16_3_15 = nativeBindings.__piNativesV16_3_15; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 45c07ec2f..183a3876d 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.3.14", + "version": "16.3.15", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index 11b7bc0d4..88b2c10be 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.3.14", + "version": "16.3.15", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index a20592455..329574cb8 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.3.14", + "version": "16.3.15", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index f5fee963b..0e71287c1 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.3.14", + "version": "16.3.15", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/package.json b/packages/tui/package.json index e672194e4..fea5e808d 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.3.14", + "version": "16.3.15", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index ccaabb9a1..2e186b008 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.3.14", + "version": "16.3.15", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index 1245b516e..51c90f13f 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.3.14", + "version": "16.3.15", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From e82592889429388e1bcf8f0ec504601482be0432 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:51:42 +0200 Subject: [PATCH 018/205] fix(ai): categorized pro-lite plans as paid in codex tier classifier - Added pro-lite plan identification to the openAI codex tier classifier. - Ensured prolite and pro_lite plan types are correctly categorized as paid. --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/auth-storage.ts | 2 ++ 2 files changed, 6 insertions(+) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 03dd313da..db588e153 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- Recognized Pro Lite as a paid plan tier for OpenAI Codex models + ## [16.3.15] - 2026-07-09 ### Breaking Changes diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 2c094515b..fb9020f0d 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -817,6 +817,8 @@ function getUsagePlanType(report: UsageReport | null): string | undefined { function classifyOpenAICodexPlan(report: UsageReport | null): OpenAICodexPlanClass { const planType = getUsagePlanType(report); if (!planType) return "unknown"; + // Pro Lite is a paid Codex tier, but does not imply full Pro-only model access. + if (planType === "prolite" || planType === "pro_lite") return "paid"; const tokens = planType.split("_"); if (tokens.some(token => OPENAI_CODEX_PRO_PLAN_TOKENS[token] === true)) return "pro"; if (tokens.some(token => OPENAI_CODEX_PAID_PLAN_TOKENS[token] === true)) return "paid"; From f21d2385e4528ad11a42385d638bb29d5364fc27 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 20:58:01 +0000 Subject: [PATCH 019/205] fix(session): skipped collapsed snapcompact frames Avoided reattaching snapcompact archive image blocks when rebuilding collapsed transcript contexts so live TUI resumes do not retain archived frames. Added regression coverage for collapsed transcripts while preserving full transcript and provider context frame reattachment. Fixes #4979 --- packages/coding-agent/CHANGELOG.md | 4 + .../src/session/session-context.test.ts | 83 +++++++++++++++++++ .../src/session/session-context.ts | 1 + 3 files changed, 88 insertions(+) create mode 100644 packages/coding-agent/src/session/session-context.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 78d22734a..25a558ec6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) + ## [16.3.14] - 2026-07-09 ### Fixed diff --git a/packages/coding-agent/src/session/session-context.test.ts b/packages/coding-agent/src/session/session-context.test.ts new file mode 100644 index 000000000..23139093a --- /dev/null +++ b/packages/coding-agent/src/session/session-context.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, it } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import * as snapcompact from "@oh-my-pi/snapcompact"; +import type { CompactionSummaryMessage } from "./messages"; +import { buildSessionContext } from "./session-context"; +import type { SessionEntry } from "./session-entries"; + +const timestamp = "2026-07-09T00:00:00.000Z"; + +const compactedEntries = [ + { + type: "message", + id: "m1", + parentId: null, + timestamp, + message: { role: "user", content: [{ type: "text", text: "before compaction" }], timestamp: 1 }, + }, + { + type: "compaction", + id: "c1", + parentId: "m1", + timestamp, + summary: "summary", + firstKeptEntryId: "m1", + tokensBefore: 123, + preserveData: { + [snapcompact.PRESERVE_KEY]: { + frames: [{ data: "base64-frame", mimeType: "image/png", cols: 10, rows: 10, chars: 100 }], + totalChars: 100, + truncatedChars: 0, + textHead: "head", + textTail: "tail", + }, + }, + }, + { + type: "message", + id: "m2", + parentId: "c1", + timestamp, + message: { role: "user", content: [{ type: "text", text: "after compaction" }], timestamp: 2 }, + }, +] satisfies SessionEntry[]; + +function compactionSummary(messages: AgentMessage[]): CompactionSummaryMessage { + const summary = messages.find( + (message): message is CompactionSummaryMessage => message.role === "compactionSummary", + ); + if (!summary) throw new Error("Expected a compaction summary message"); + return summary; +} + +describe("buildSessionContext snapcompact archives", () => { + it("omits snapcompact archive blocks from collapsed transcript summaries", () => { + const context = buildSessionContext(compactedEntries, undefined, undefined, { + transcript: true, + collapseCompactedHistory: true, + }); + + const summary = compactionSummary(context.messages); + + expect(summary.images).toBeUndefined(); + expect(summary.blocks).toBeUndefined(); + }); + + it("keeps snapcompact archive blocks in full transcript summaries", () => { + const context = buildSessionContext(compactedEntries, undefined, undefined, { transcript: true }); + + const summary = compactionSummary(context.messages); + + expect(summary.images?.map(image => image.data)).toEqual(["base64-frame"]); + expect(summary.blocks?.map(block => block.type)).toEqual(["text", "image", "text"]); + }); + + it("keeps snapcompact archive blocks in provider context summaries", () => { + const context = buildSessionContext(compactedEntries); + + const summary = compactionSummary(context.messages); + + expect(summary.images?.map(image => image.data)).toEqual(["base64-frame"]); + expect(summary.blocks?.map(block => block.type)).toEqual(["text", "image", "text"]); + }); +}); diff --git a/packages/coding-agent/src/session/session-context.ts b/packages/coding-agent/src/session/session-context.ts index 769f92933..c17f06f32 100644 --- a/packages/coding-agent/src/session/session-context.ts +++ b/packages/coding-agent/src/session/session-context.ts @@ -132,6 +132,7 @@ function snapcompactHistoryBlocksForContext( options: BuildSessionContextOptions | undefined, ) { if (!archive) return undefined; + if (options?.transcript && options.collapseCompactedHistory) return undefined; return snapcompact.historyBlocks(archive, snapcompactHistoryBlockOptions(archive, options)); } From c143185c018e98812d44d365c0710fa5006f5d20 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 21:03:25 +0000 Subject: [PATCH 020/205] fix(ai): rechecked codex blocks during selection - Re-fetched usage for blocked Codex OAuth candidates during ranking so fresh recovered windows can clear stale persisted blocks. - Relaxed Codex block reconciliation to trust live allowed/limitReached metadata with an available primary window. - Added regression coverage for selection-path stale block recovery. Fixes #4980 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/auth-storage.ts | 51 +++++++---- .../test/auth-storage-codex-selection.test.ts | 91 +++++++++++++++---- 3 files changed, 112 insertions(+), 34 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index e6d81932f..a74b0c89c 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks when live usage shows the 5-hour window recovered ([#4980](https://github.com/can1357/oh-my-pi/issues/4980)). + ## [16.3.14] - 2026-07-09 ### Changed diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 315d43c48..886052d4a 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3316,18 +3316,36 @@ export class AuthStorage { args.order.map(async idx => { const selection = args.credentials[idx]; if (!selection) return null; - const blockedUntil = this.#getCredentialBlockedUntil( + let blockedUntil = this.#getCredentialBlockedUntil( args.provider, args.providerKey, selection.index, args.blockScope, ); - if (blockedUntil !== undefined) return { selection, usage: null, usageChecked: false, blockedUntil }; - const usage = await this.#getUsageReport(args.provider, selection.credential, { - ...args.options, - timeoutMs: this.#usageRequestTimeoutMs, - }); - return { selection, usage, usageChecked: true, blockedUntil: undefined as number | undefined }; + let usage: UsageReport | null = null; + let usageChecked = false; + if (blockedUntil !== undefined && args.provider === "openai-codex") { + usage = await this.#getUsageReport(args.provider, selection.credential, { + ...args.options, + timeoutMs: this.#usageRequestTimeoutMs, + }); + usageChecked = true; + blockedUntil = this.#getCredentialBlockedUntil( + args.provider, + args.providerKey, + selection.index, + args.blockScope, + ); + } + if (blockedUntil !== undefined) return { selection, usage, usageChecked, blockedUntil }; + if (!usageChecked) { + usage = await this.#getUsageReport(args.provider, selection.credential, { + ...args.options, + timeoutMs: this.#usageRequestTimeoutMs, + }); + usageChecked = true; + } + return { selection, usage, usageChecked, blockedUntil: undefined as number | undefined }; }), ); const timeoutSignal = Promise.withResolvers(); @@ -4371,19 +4389,18 @@ export class AuthStorage { /** * Self-heal a stale Codex usage-limit block: when a fresh live usage report - * shows the account is allowed and below every limit, drop the persisted and - * in-memory `openai-codex:oauth` blocks so the balancer re-includes it. Only - * Codex — its ranking strategy uses the single model-independent `"shared"` - * scope, so clearing every block for the credential id is exact. + * says the account is allowed and its primary rolling window is available, + * drop the persisted and in-memory `openai-codex:oauth` blocks so credential + * selection can re-include seats whose short window recovered before a + * longer persisted block naturally expires. */ #isHealthyCodexUsageReport(report: UsageReport): boolean { + if (report.provider !== "openai-codex") return false; const metadata = report.metadata; - return ( - report.provider === "openai-codex" && - metadata?.allowed === true && - metadata.limitReached === false && - !this.#isUsageLimitReached(report.limits) - ); + if (metadata?.allowed !== true || metadata.limitReached !== false) return false; + const primary = codexRankingStrategy.findWindowLimits(report).primary; + if (primary) return !this.#isUsageLimitExhausted(primary); + return !this.#isUsageLimitReached(report.limits); } #reconcileCodexUsageBlockForCredential(provider: Provider, credentialId: number, report: UsageReport): void { diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6bf0ff584..a63ac005e 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -385,15 +385,6 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const blockedSelectionCounts = await countApiKeySelections( - authStorage, - "openai-codex", - "stale-codex-block-before-fetch", - 40, - ); - expect(countFor(blockedSelectionCounts, "api-acct-blocked")).toBe(0); - expect(countFor(blockedSelectionCounts, "api-acct-healthy")).toBeGreaterThan(0); - const generationBeforeFetch = authStorage.getGeneration(); await authStorage.fetchUsageReports(); @@ -411,6 +402,80 @@ describe("AuthStorage codex oauth ranking", () => { expect(countFor(reconciledSelectionCounts, "api-acct-healthy")).toBeGreaterThan(0); }); + test("re-evaluates a stale persisted Codex block during selection when the 5h window recovered", async () => { + if (!authStorage || !store?.upsertCredentialBlock || !store.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-recovered-blocked", "recovered-blocked@example.com") }, + { type: "oauth", ...createCredential("acct-recovered-sibling", "recovered-sibling@example.com") }, + ]); + + const blockedRow = store.listAuthCredentials("openai-codex").find(row => { + const credential = row.credential; + return credential.type === "oauth" && credential.accountId === "acct-recovered-blocked"; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + + store.upsertCredentialBlock({ + credentialId: blockedRow.id, + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, + }); + + const fiveHourWindow: UsageWindowConfig = { + windowId: "5h", + windowLabel: "5 Hours", + durationMs: FIVE_HOUR_MS, + }; + + usageByAccount.set( + "acct-recovered-blocked", + createCodexUsageReport({ + accountId: "acct-recovered-blocked", + primary: { usedFraction: 0.2, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: 6 * 24 * HOUR_MS }, + primaryWindow: fiveHourWindow, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "recovered-blocked@example.com", + accountId: "acct-recovered-blocked", + }, + }), + ); + usageByAccount.set( + "acct-recovered-sibling", + createCodexUsageReport({ + accountId: "acct-recovered-sibling", + primary: { usedFraction: 0.2, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: 6 * 24 * HOUR_MS }, + primaryWindow: fiveHourWindow, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "recovered-sibling@example.com", + accountId: "acct-recovered-sibling", + }, + }), + ); + + const selectionCounts = await countApiKeySelections( + authStorage, + "openai-codex", + "codex-stale-block-selection-recovered", + 150, + ); + + expect(countFor(selectionCounts, "api-acct-recovered-blocked")).toBeGreaterThan(0); + expect(countFor(selectionCounts, "api-acct-recovered-sibling")).toBeGreaterThan(0); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeUndefined(); + }); + test("an older in-flight healthy Codex usage report does not clear a newer usage-limit block", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); @@ -496,14 +561,6 @@ describe("AuthStorage codex oauth ranking", () => { await inFlightReports; expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); - const selectionCounts = await countApiKeySelections( - authStorage, - "openai-codex", - "codex-inflight-race-after-resolve", - 40, - ); - expect(countFor(selectionCounts, "api-acct-race-blocked")).toBe(0); - expect(countFor(selectionCounts, "api-acct-race-healthy")).toBeGreaterThan(0); }); test("broker-sourced healthy Codex usage clears remote gateway backoff", async () => { From 5b0d670d6b901a0140b74e6f9a8532807f9e9c44 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 21:17:08 +0000 Subject: [PATCH 021/205] fix(ai): kept exhausted codex blocks - Restored all-limit Codex block clearing so recovered primary windows do not clear blocks while another reported quota remains exhausted. - Added regression coverage for primary-recovered/secondary-exhausted selection-path reconciliation. Fixes #4980 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/auth-storage.ts | 9 +-- .../test/auth-storage-codex-selection.test.ts | 75 +++++++++++++++++++ 3 files changed, 79 insertions(+), 7 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index a74b0c89c..417ccf1ff 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks when live usage shows the 5-hour window recovered ([#4980](https://github.com/can1357/oh-my-pi/issues/4980)). +- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks when live usage shows all reported windows recovered ([#4980](https://github.com/can1357/oh-my-pi/issues/4980)). ## [16.3.14] - 2026-07-09 diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 886052d4a..bd7b42fb6 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -4389,17 +4389,14 @@ export class AuthStorage { /** * Self-heal a stale Codex usage-limit block: when a fresh live usage report - * says the account is allowed and its primary rolling window is available, - * drop the persisted and in-memory `openai-codex:oauth` blocks so credential - * selection can re-include seats whose short window recovered before a - * longer persisted block naturally expires. + * says the account is allowed and below every reported limit, drop the + * persisted and in-memory `openai-codex:oauth` blocks so credential selection + * can re-include recovered seats before a stale block naturally expires. */ #isHealthyCodexUsageReport(report: UsageReport): boolean { if (report.provider !== "openai-codex") return false; const metadata = report.metadata; if (metadata?.allowed !== true || metadata.limitReached !== false) return false; - const primary = codexRankingStrategy.findWindowLimits(report).primary; - if (primary) return !this.#isUsageLimitExhausted(primary); return !this.#isUsageLimitReached(report.limits); } diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index a63ac005e..14d0e7c8a 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -476,6 +476,81 @@ describe("AuthStorage codex oauth ranking", () => { expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeUndefined(); }); + test("keeps a stale Codex block when the 5h window recovered but the 7d window remains exhausted", async () => { + if (!authStorage || !store?.upsertCredentialBlock || !store.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-secondary-exhausted", "secondary-exhausted@example.com") }, + { type: "oauth", ...createCredential("acct-secondary-healthy", "secondary-healthy@example.com") }, + ]); + + const blockedRow = store.listAuthCredentials("openai-codex").find(row => { + const credential = row.credential; + return credential.type === "oauth" && credential.accountId === "acct-secondary-exhausted"; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + + const blockedUntilMs = Date.now() + 6 * 24 * HOUR_MS; + store.upsertCredentialBlock({ + credentialId: blockedRow.id, + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs, + }); + + const fiveHourWindow: UsageWindowConfig = { + windowId: "5h", + windowLabel: "5 Hours", + durationMs: FIVE_HOUR_MS, + }; + + usageByAccount.set( + "acct-secondary-exhausted", + createCodexUsageReport({ + accountId: "acct-secondary-exhausted", + primary: { usedFraction: 0.2, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 1, resetInMs: 6 * 24 * HOUR_MS }, + primaryWindow: fiveHourWindow, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "secondary-exhausted@example.com", + accountId: "acct-secondary-exhausted", + }, + }), + ); + usageByAccount.set( + "acct-secondary-healthy", + createCodexUsageReport({ + accountId: "acct-secondary-healthy", + primary: { usedFraction: 0.2, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: 6 * 24 * HOUR_MS }, + primaryWindow: fiveHourWindow, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "secondary-healthy@example.com", + accountId: "acct-secondary-healthy", + }, + }), + ); + + const selectionCounts = await countApiKeySelections( + authStorage, + "openai-codex", + "codex-stale-block-secondary-exhausted", + 150, + ); + + expect(countFor(selectionCounts, "api-acct-secondary-exhausted")).toBe(0); + expect(countFor(selectionCounts, "api-acct-secondary-healthy")).toBeGreaterThan(0); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBe(blockedUntilMs); + }); + test("an older in-flight healthy Codex usage report does not clear a newer usage-limit block", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); From 0ffc43bfc09b1d784c2e9afa261d32ee8d75dcfd Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 21:36:35 +0000 Subject: [PATCH 022/205] fix(ai): preserved fresh codex blocks - Delayed healthy-usage reconciliation for newly-set local Codex blocks so lagging /usage responses cannot immediately undo a real 429 backoff. - Added regression coverage for fresh usage-limit blocks that see healthy usage during selection. - Kept broker stale-block reconciliation covered by seeding a persisted-only block. Fixes #4980 --- packages/ai/src/auth-storage.ts | 20 +++ .../test/auth-storage-codex-selection.test.ts | 114 +++++++++++++----- 2 files changed, 103 insertions(+), 31 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index bd7b42fb6..acdaff5e7 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -965,6 +965,8 @@ export class AuthStorage { #sessionLastCredential: Map> = new Map(); /** Maps provider:type -> credentialIndex -> blockedUntilMs for temporary backoff. */ #credentialBackoff: Map> = new Map(); + /** Earliest time a freshly-set in-memory block may be cleared by live usage reconciliation. */ + #credentialBackoffProbeAfter: Map> = new Map(); #usageProviderResolver?: (provider: Provider) => UsageProvider | undefined; #rankingStrategyResolver?: (provider: Provider) => CredentialRankingStrategy | undefined; #usageCache: UsageCache; @@ -1345,6 +1347,9 @@ export class AuthStorage { if (backoffMap.size === 0) { this.#credentialBackoff.delete(backoffKey); } + const probeAfterMap = this.#credentialBackoffProbeAfter.get(backoffKey); + probeAfterMap?.delete(credentialIndex); + if (probeAfterMap?.size === 0) this.#credentialBackoffProbeAfter.delete(backoffKey); return undefined; } return blockedUntil; @@ -1437,6 +1442,9 @@ export class AuthStorage { const nextBlockedUntil = Math.max(existing, blockedUntilMs); backoffMap.set(credentialIndex, nextBlockedUntil); this.#credentialBackoff.set(backoffKey, backoffMap); + const probeAfterMap = this.#credentialBackoffProbeAfter.get(backoffKey) ?? new Map(); + probeAfterMap.set(credentialIndex, Math.min(nextBlockedUntil, Date.now() + USAGE_REPORT_TTL_MS)); + this.#credentialBackoffProbeAfter.set(backoffKey, probeAfterMap); this.#invalidateUsageReportCache(provider); const upsertCredentialBlock = this.#store.upsertCredentialBlock?.bind(this.#store); @@ -4385,6 +4393,11 @@ export class AuthStorage { backoffMap.delete(index); if (backoffMap.size === 0) this.#credentialBackoff.delete(key); } + for (const [key, probeAfterMap] of this.#credentialBackoffProbeAfter) { + if (key !== providerKey && !key.startsWith(scopedPrefix)) continue; + probeAfterMap.delete(index); + if (probeAfterMap.size === 0) this.#credentialBackoffProbeAfter.delete(key); + } } /** @@ -4410,6 +4423,13 @@ export class AuthStorage { const blockScope = this.#rankingStrategyResolver?.(provider)?.blockScope?.({}); const blockedUntilMs = this.#getCredentialBlockedUntil(provider, providerKey, credentialIndex, blockScope); if (blockedUntilMs === undefined) return; + // `/usage` can lag the request path that just returned 429. Fresh local + // blocks get one usage-cache window before healthy reports may clear them. + const nowMs = Date.now(); + const scopedBackoffKey = this.#toScopedBackoffKey(providerKey, blockScope); + const globalProbeAfterMs = this.#credentialBackoffProbeAfter.get(providerKey)?.get(credentialIndex) ?? 0; + const scopedProbeAfterMs = this.#credentialBackoffProbeAfter.get(scopedBackoffKey)?.get(credentialIndex) ?? 0; + if (Math.max(globalProbeAfterMs, scopedProbeAfterMs) > nowMs) return; this.#clearCredentialBlocks(provider, credentialId); logger.info("Cleared stale Codex usage-limit block after healthy live usage report", { credentialId, diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 14d0e7c8a..3963880e0 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -551,6 +551,73 @@ describe("AuthStorage codex oauth ranking", () => { expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBe(blockedUntilMs); }); + test("keeps a fresh Codex usage-limit block when selection sees healthy usage", async () => { + if (!authStorage || !store?.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-fresh-blocked", "fresh-blocked@example.com") }, + { type: "oauth", ...createCredential("acct-fresh-healthy", "fresh-healthy@example.com") }, + ]); + + usageByAccount.set( + "acct-fresh-blocked", + createCodexUsageReport({ + accountId: "acct-fresh-blocked", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "fresh-blocked@example.com", + accountId: "acct-fresh-blocked", + }, + }), + ); + usageByAccount.set( + "acct-fresh-healthy", + createCodexUsageReport({ + accountId: "acct-fresh-healthy", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "fresh-healthy@example.com", + accountId: "acct-fresh-healthy", + }, + }), + ); + + const blockedRow = store.listAuthCredentials("openai-codex").find(row => { + const credential = row.credential; + return credential.type === "oauth" && credential.accountId === "acct-fresh-blocked"; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + + let blockedSessionId: string | undefined; + for (let index = 0; index < 100; index += 1) { + const sessionId = `codex-fresh-block-selected-${index}`; + if ((await authStorage.getApiKey("openai-codex", sessionId)) === "api-acct-fresh-blocked") { + blockedSessionId = sessionId; + break; + } + } + if (!blockedSessionId) throw new Error("expected a session selecting the soon-blocked account"); + + const markResult = await authStorage.markUsageLimitReached("openai-codex", blockedSessionId, { + retryAfterMs: 6 * 24 * HOUR_MS, + }); + + expect(markResult.switched).toBe(true); + const selectionAfterBlock = await authStorage.getApiKey("openai-codex", blockedSessionId); + expect(selectionAfterBlock).not.toBe("api-acct-fresh-blocked"); + expect(selectionAfterBlock).toBe("api-acct-fresh-healthy"); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + }); test("an older in-flight healthy Codex usage report does not clear a newer usage-limit block", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); @@ -639,7 +706,7 @@ describe("AuthStorage codex oauth ranking", () => { }); test("broker-sourced healthy Codex usage clears remote gateway backoff", async () => { - if (!authStorage || !store?.getCredentialBlock) { + if (!authStorage || !store?.getCredentialBlock || !store.upsertCredentialBlock) { throw new Error("test setup failed"); } @@ -679,6 +746,18 @@ describe("AuthStorage codex oauth ranking", () => { }), ); + const staleBlockedRow = store.listAuthCredentials("openai-codex").find(row => { + const credential = row.credential; + return credential.type === "oauth" && credential.accountId === "acct-broker-blocked"; + }); + if (!staleBlockedRow) throw new Error("expected stale blocked credential row"); + store.upsertCredentialBlock({ + credentialId: staleBlockedRow.id, + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, + }); + const token = "codex-broker-reconcile"; const handle = startAuthBroker({ storage: authStorage, @@ -688,18 +767,6 @@ describe("AuthStorage codex oauth ranking", () => { }); try { const brokerClient = new AuthBrokerClient({ url: handle.url, token }); - const originalUpsertCredentialBlock = brokerClient.upsertCredentialBlock.bind(brokerClient); - const blockPersisted = Promise.withResolvers(); - vi.spyOn(brokerClient, "upsertCredentialBlock").mockImplementation(async (id, block, signal) => { - try { - const response = await originalUpsertCredentialBlock(id, block, signal); - blockPersisted.resolve(); - return response; - } catch (error) { - blockPersisted.reject(error); - throw error; - } - }); const initialResult = await brokerClient.fetchSnapshot(); if (initialResult.status !== 200) throw new Error("expected broker snapshot"); const blockedRow = initialResult.snapshot.credentials.find(entry => { @@ -715,31 +782,16 @@ describe("AuthStorage codex oauth ranking", () => { const clientStorage = new AuthStorage(remoteStore); await clientStorage.reload(); try { - let blockedSessionId: string | undefined; - for (let index = 0; index < 100; index += 1) { - const sessionId = `broker-codex-local-block-${index}`; - const apiKey = await clientStorage.getApiKey("openai-codex", sessionId); - if (apiKey === "api-acct-broker-blocked") { - blockedSessionId = sessionId; - break; - } - } - if (!blockedSessionId) throw new Error("expected a session selecting the blocked account"); - - const markResult = await clientStorage.markUsageLimitReached("openai-codex", blockedSessionId, { - retryAfterMs: 6 * 24 * HOUR_MS, - }); - - expect(markResult.switched).toBe(true); expect(remoteStore.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); - await blockPersisted.promise; expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); await clientStorage.fetchUsageReports(); expect(remoteStore.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeUndefined(); expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeUndefined(); - expect(await clientStorage.getApiKey("openai-codex", blockedSessionId)).toBe("api-acct-broker-blocked"); + expect(await clientStorage.getApiKey("openai-codex", "broker-codex-reconciled")).toBe( + "api-acct-broker-blocked", + ); } finally { clientStorage.close(); remoteStore.close(); From bf6480647491c8fbd7eed960e4506a8c6612b377 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 21:39:42 +0000 Subject: [PATCH 023/205] fix(mcp): kept macos stdio servers attached Left Darwin stdio MCP server launches in the inherited session so macOS TCC can prompt for Apple Events permissions used by xcrun mcpbridge. Added resolver coverage for Darwin while preserving Linux detach and Windows console behavior. Fixes #4987 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../src/mcp/transports/stdio.test.ts | 12 ++++++++++ .../coding-agent/src/mcp/transports/stdio.ts | 22 ++++++++++++------- .../src/modes/controllers/input-controller.ts | 8 +++---- 4 files changed, 34 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..5736aabe7 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed macOS stdio MCP servers launching in a detached session, so `xcrun mcpbridge` can trigger the TCC Apple Events permission prompt and complete startup. ([#4987](https://github.com/can1357/oh-my-pi/issues/4987)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/mcp/transports/stdio.test.ts b/packages/coding-agent/src/mcp/transports/stdio.test.ts index 57a2d161e..ac1364787 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.test.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.test.ts @@ -31,6 +31,18 @@ describe("resolveStdioSpawnCommand", () => { }); }); + it("keeps Darwin stdio MCP servers attached so TCC Apple Events prompts can resolve", async () => { + await expect( + resolveStdioSpawnCommand( + { command: "xcrun", args: ["mcpbridge"] }, + { cwd: process.cwd(), env: {}, platform: "darwin" }, + ), + ).resolves.toEqual({ + cmd: ["xcrun", "mcpbridge"], + detached: false, + }); + }); + it("detaches off-Windows MCP servers so terminal job-control signals cannot stop them", async () => { await expect( resolveStdioSpawnCommand( diff --git a/packages/coding-agent/src/mcp/transports/stdio.ts b/packages/coding-agent/src/mcp/transports/stdio.ts index f9d4e04e0..228226008 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.ts @@ -37,13 +37,18 @@ export interface StdioSpawnCommand { */ windowsHide?: boolean; /** - * Run the subprocess in its own session. + * Run the subprocess in its own session when the platform can safely do so. * - * POSIX: `true`. Detach → `setsid`, so the MCP process tree has no - * controlling terminal and terminal job-control signals (Ctrl+Z SIGTSTP, + * Linux/other POSIX: `true`. Detach → `setsid`, so the MCP process tree has + * no controlling terminal and terminal job-control signals (Ctrl+Z SIGTSTP, * background-read SIGTTIN) cannot stop stdio servers such as * `chrome-devtools-mcp` and leave our read loop blocked on silent pipes. * + * macOS: `false`. LaunchServices/TCC attributes Apple Events automation to + * the responsible terminal process only while the child stays in the + * inherited session; detaching via `setsid` prevents the permission prompt + * for servers such as `xcrun mcpbridge` (#4987). + * * Windows: `false`. There is no SIGTSTP/SIGTTIN to escape, and Windows * wrapper chains must stay in the OMP console session so nested console * grandchildren keep stdout routed through our pipe (#3544). @@ -247,7 +252,7 @@ export async function resolveStdioSpawnCommand( options: ResolveStdioSpawnOptions, ): Promise { const args = config.args ?? []; - if (options.platform !== "win32") return { cmd: [config.command, ...args], detached: true }; + if (options.platform !== "win32") return { cmd: [config.command, ...args], detached: options.platform !== "darwin" }; const windowsHide = options.hostHasInheritableConsole === undefined ? true : !options.hostHasInheritableConsole; const resolved = await resolveWindowsCommandPath(config.command, options.cwd, options.env); @@ -366,10 +371,11 @@ export class StdioTransport implements MCPTransport { }); // Platform-derived session and console-window handling come from - // `resolveStdioSpawnCommand`: POSIX detaches into its own session to - // escape terminal job-control signals (SIGTSTP, SIGTTIN); Windows stays - // attached, and only hides the child when the host has no console to - // share. See `StdioSpawnCommand`. + // `resolveStdioSpawnCommand`: Linux/other POSIX detach into their own + // session to escape terminal job-control signals (SIGTSTP, SIGTTIN); + // macOS stays attached so TCC can prompt for Apple Events automation; + // Windows stays attached, and only hides the child when the host has no + // console to share. See `StdioSpawnCommand`. this.#process = spawn({ cmd: spawnCommand.cmd, cwd, diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index bb19390b4..f5aa2766a 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -1024,10 +1024,10 @@ export class InputController { // leaves wrappers and pipeline peers running and the terminal // hung — exactly the failure shape we're fixing. Stopping the whole // group keeps the shell's job-control view consistent. Long-lived - // children that must survive the suspend (MCP stdio servers via - // the `detached: true` spawn in `mcp/transports/stdio.ts`, every - // brush external command via brush's per-child `setsid` in - // `crates/vendor/brush-core/src/commands.rs`) are already in + // children that must survive the suspend (Linux/other POSIX MCP stdio + // servers via the platform-specific `detached: true` spawn in + // `mcp/transports/stdio.ts`, every brush external command via brush's + // per-child `setsid` in `crates/vendor/brush-core/src/commands.rs`) are // their own sessions, so pgid=0 does not reach them. process.kill(0, "SIGSTOP"); } catch (err) { From 04214a8c8504dd1e391745dfe2599314126fabc9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 22:07:08 +0000 Subject: [PATCH 024/205] fix(ai): guarded broker codex blocks - Preserved fresh broker-sourced credential blocks by exposing store-level reconciliation delays to AuthStorage. - Tracked fresh block observations in SQLite and remote broker stores without protecting initial stale snapshots. - Added broker sibling regression coverage for healthy usage lag after a shared 429 block. Fixes #4980 --- packages/ai/src/auth-broker/remote-store.ts | 52 +++++++- packages/ai/src/auth-storage.ts | 30 ++++- .../test/auth-storage-codex-selection.test.ts | 120 ++++++++++++++++++ 3 files changed, 195 insertions(+), 7 deletions(-) diff --git a/packages/ai/src/auth-broker/remote-store.ts b/packages/ai/src/auth-broker/remote-store.ts index f56ba6b3a..a2c30eb61 100644 --- a/packages/ai/src/auth-broker/remote-store.ts +++ b/packages/ai/src/auth-broker/remote-store.ts @@ -39,6 +39,7 @@ import type { * one broker call instead of N. */ const USAGE_CACHE_TTL_MS = 15_000; +const CREDENTIAL_BLOCK_RECONCILE_DELAY_MS = 5 * 60_000; const WAIT_THRESHOLD_MS = 1_000; const MAX_WAIT_MS = 5_000; const BACKGROUND_WAIT_MS = 30_000; @@ -208,6 +209,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { #cache: Map = new Map(); #usageCache?: UsageCacheEntry; #usageInflight?: Promise; + #credentialBlockReconcileAfter: Map = new Map(); #usageCacheEpoch = 0; #closed = false; /** @@ -223,7 +225,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { constructor(opts: RemoteAuthCredentialStoreOptions) { this.#client = opts.client; this.#streamSnapshots = opts.streamSnapshots ?? true; - this.#applySnapshot(opts.initialSnapshot ?? emptySnapshot(), opts.initialSnapshot?.generation ?? 0); + this.#applySnapshot(opts.initialSnapshot ?? emptySnapshot(), opts.initialSnapshot?.generation ?? 0, false); this.#onSnapshot = opts.onSnapshot; void this.#runBackground(); } @@ -236,10 +238,12 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { return this.#snapshot; } - #applySnapshot(snapshot: SnapshotResponse, generation: number): void { + #applySnapshot(snapshot: SnapshotResponse, generation: number, protectNewBlocks = true): void { const nowMs = Date.now(); + const previousCredentials = this.#snapshot.credentials; const credentials = snapshot.credentials.map(entry => this.#normalizeSnapshotEntryBlocks(entry, nowMs)); - if (snapshotBlocksChanged(this.#snapshot.credentials, credentials)) this.#invalidateUsageCache(); + if (snapshotBlocksChanged(previousCredentials, credentials)) this.#invalidateUsageCache(); + if (protectNewBlocks) this.#protectNewSnapshotBlocks(previousCredentials, credentials, nowMs); this.#snapshot = { ...snapshot, credentials }; this.#generation = generation; this.#snapshotReceivedAt = nowMs; @@ -251,6 +255,29 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { logger.debug("auth-broker snapshot callback failed", { error: String(error) }); } } + #protectNewSnapshotBlocks(previous: readonly SnapshotEntry[], next: readonly SnapshotEntry[], nowMs: number): void { + const previousBlocksByKey = new Map(); + for (const entry of previous) { + for (const block of entry.blocks ?? []) { + previousBlocksByKey.set(`${entry.id}\0${block.providerKey}\0${block.blockScope}`, block.blockedUntilMs); + } + } + const activeKeys = new Set(); + for (const entry of next) { + for (const block of entry.blocks ?? []) { + const key = `${entry.id}\0${block.providerKey}\0${block.blockScope}`; + activeKeys.add(key); + if (previousBlocksByKey.get(key) === block.blockedUntilMs) continue; + this.#credentialBlockReconcileAfter.set( + key, + Math.min(block.blockedUntilMs, nowMs + CREDENTIAL_BLOCK_RECONCILE_DELAY_MS), + ); + } + } + for (const key of this.#credentialBlockReconcileAfter.keys()) { + if (!activeKeys.has(key)) this.#credentialBlockReconcileAfter.delete(key); + } + } async #runBackground(): Promise { let backoffMs = BACKGROUND_BACKOFF_INITIAL_MS; @@ -340,11 +367,13 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { const incoming = this.#normalizeSnapshotEntryBlocks(entry, Date.now()); const index = this.#snapshot.credentials.findIndex(candidate => candidate.id === incoming.id); const previousBlocks = index === -1 ? undefined : this.#snapshot.credentials[index]?.blocks; - if (!credentialBlockSnapshotsEqual(previousBlocks, incoming.blocks)) this.#invalidateUsageCache(); + const blocksChanged = !credentialBlockSnapshotsEqual(previousBlocks, incoming.blocks); + if (blocksChanged) this.#invalidateUsageCache(); const credentials = index === -1 ? [...this.#snapshot.credentials, incoming] : this.#snapshot.credentials.map((candidate, i) => (i === index ? incoming : candidate)); + if (blocksChanged) this.#protectNewSnapshotBlocks(this.#snapshot.credentials, credentials, Date.now()); this.#snapshot = { ...this.#snapshot, generation, serverNowMs, refresher, credentials }; this.#generation = generation; this.#snapshotReceivedAt = Date.now(); @@ -392,6 +421,11 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { return block.blockedUntilMs; } + getCredentialBlockReconcileAfter(credentialId: number, providerKey: string, blockScope: string): number | undefined { + if (this.getCredentialBlock(credentialId, providerKey, blockScope) === undefined) return undefined; + return this.#credentialBlockReconcileAfter.get(`${credentialId}\0${providerKey}\0${blockScope}`); + } + listCredentialBlocks(credentialIds: readonly number[]): StoredCredentialBlock[] { const nowMs = Date.now(); this.cleanExpiredCredentialBlocks(nowMs); @@ -416,6 +450,10 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { upsertCredentialBlock(block: StoredCredentialBlock): void { this.#upsertSnapshotBlock(block); this.#invalidateUsageCache(); + this.#credentialBlockReconcileAfter.set( + `${block.credentialId}\0${block.providerKey}\0${block.blockScope}`, + Math.min(block.blockedUntilMs, Date.now() + CREDENTIAL_BLOCK_RECONCILE_DELAY_MS), + ); const body = toCredentialBlockSnapshot(block); void this.#client .upsertCredentialBlock(block.credentialId, body) @@ -434,6 +472,9 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { deleteCredentialBlocks(credentialId: number): void { this.#deleteSnapshotBlocks(credentialId); + for (const key of this.#credentialBlockReconcileAfter.keys()) { + if (key.startsWith(`${credentialId}\0`)) this.#credentialBlockReconcileAfter.delete(key); + } this.#invalidateUsageCache(); void this.#client .deleteCredentialBlocks(credentialId) @@ -450,6 +491,9 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { cleanExpiredCredentialBlocks(nowMs: number): void { this.#pruneExpiredCredentialBlocks(nowMs); + for (const [key, reconcileAfterMs] of this.#credentialBlockReconcileAfter) { + if (reconcileAfterMs <= nowMs) this.#credentialBlockReconcileAfter.delete(key); + } } /** diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index acdaff5e7..36eb052d9 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -320,6 +320,8 @@ export interface AuthCredentialStore { cleanExpiredCache(): void; /** Non-expired block for one (credential, providerKey, scope) key, or undefined. */ getCredentialBlock?(credentialId: number, providerKey: string, blockScope: string): number | undefined; + /** Earliest time a shared-store block should be eligible for live-usage reconciliation. */ + getCredentialBlockReconcileAfter?(credentialId: number, providerKey: string, blockScope: string): number | undefined; /** Upsert with MAX semantics: keep the later blockedUntilMs on conflict. */ upsertCredentialBlock?(block: StoredCredentialBlock): void; /** Drop every block row for a credential (all providerKeys/scopes). */ @@ -4423,13 +4425,19 @@ export class AuthStorage { const blockScope = this.#rankingStrategyResolver?.(provider)?.blockScope?.({}); const blockedUntilMs = this.#getCredentialBlockedUntil(provider, providerKey, credentialIndex, blockScope); if (blockedUntilMs === undefined) return; - // `/usage` can lag the request path that just returned 429. Fresh local - // blocks get one usage-cache window before healthy reports may clear them. + // `/usage` can lag the request path that just returned 429. Fresh local or + // broker-sourced blocks get one usage-cache window before healthy reports may + // clear them. const nowMs = Date.now(); const scopedBackoffKey = this.#toScopedBackoffKey(providerKey, blockScope); const globalProbeAfterMs = this.#credentialBackoffProbeAfter.get(providerKey)?.get(credentialIndex) ?? 0; const scopedProbeAfterMs = this.#credentialBackoffProbeAfter.get(scopedBackoffKey)?.get(credentialIndex) ?? 0; - if (Math.max(globalProbeAfterMs, scopedProbeAfterMs) > nowMs) return; + const getStoreReconcileAfter = this.#store.getCredentialBlockReconcileAfter?.bind(this.#store); + const storeGlobalProbeAfterMs = getStoreReconcileAfter?.(credentialId, providerKey, "") ?? 0; + const storeScopedProbeAfterMs = getStoreReconcileAfter?.(credentialId, providerKey, blockScope ?? "") ?? 0; + if (Math.max(globalProbeAfterMs, scopedProbeAfterMs, storeGlobalProbeAfterMs, storeScopedProbeAfterMs) > nowMs) { + return; + } this.#clearCredentialBlocks(provider, credentialId); logger.info("Cleared stale Codex usage-limit block after healthy live usage report", { credentialId, @@ -5148,6 +5156,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { #upsertCredentialBlockStmt: Statement; #deleteCredentialBlocksStmt: Statement; #deleteExpiredCredentialBlocksStmt: Statement; + #credentialBlockReconcileAfter: Map = new Map(); #insertUsageHistoryStmt: Statement; #insertUsageCostStmt: Statement; #listUsageCostsStmt: Statement; @@ -5816,6 +5825,11 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { return typeof row?.blocked_until_ms === "number" ? row.blocked_until_ms : undefined; } + getCredentialBlockReconcileAfter(credentialId: number, providerKey: string, blockScope: string): number | undefined { + if (this.getCredentialBlock(credentialId, providerKey, blockScope) === undefined) return undefined; + return this.#credentialBlockReconcileAfter.get(`${credentialId}\0${providerKey}\0${blockScope}`); + } + upsertCredentialBlock(block: StoredCredentialBlock): void { this.#upsertCredentialBlockStmt.run( block.credentialId, @@ -5823,14 +5837,24 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { block.blockScope, block.blockedUntilMs, ); + this.#credentialBlockReconcileAfter.set( + `${block.credentialId}\0${block.providerKey}\0${block.blockScope}`, + Math.min(block.blockedUntilMs, Date.now() + USAGE_REPORT_TTL_MS), + ); } deleteCredentialBlocks(credentialId: number): void { this.#deleteCredentialBlocksStmt.run(credentialId); + for (const key of this.#credentialBlockReconcileAfter.keys()) { + if (key.startsWith(`${credentialId}\0`)) this.#credentialBlockReconcileAfter.delete(key); + } } cleanExpiredCredentialBlocks(nowMs: number): void { this.#deleteExpiredCredentialBlocksStmt.run(nowMs); + for (const [key, reconcileAfterMs] of this.#credentialBlockReconcileAfter) { + if (reconcileAfterMs <= nowMs) this.#credentialBlockReconcileAfter.delete(key); + } } listCredentialBlocks(credentialIds: readonly number[]): StoredCredentialBlock[] { diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 3963880e0..880074ad3 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -13,6 +13,7 @@ const WEEK_MS = 7 * 24 * 60 * 60 * 1000; const HOUR_MS = 60 * 60 * 1000; const FIVE_HOUR_MS = 5 * HOUR_MS; +const STALE_BLOCK_GUARD_MS = 5 * 60_000 + 1; type UsageWindowSpec = { usedFraction: number; @@ -353,6 +354,7 @@ describe("AuthStorage codex oauth ranking", () => { blockScope: "shared", blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, }); + store.cleanExpiredCredentialBlocks?.(Date.now() + STALE_BLOCK_GUARD_MS); usageByAccount.set( "acct-blocked", @@ -424,6 +426,7 @@ describe("AuthStorage codex oauth ranking", () => { blockScope: "shared", blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, }); + store.cleanExpiredCredentialBlocks?.(Date.now() + STALE_BLOCK_GUARD_MS); const fiveHourWindow: UsageWindowConfig = { windowId: "5h", @@ -618,6 +621,122 @@ describe("AuthStorage codex oauth ranking", () => { expect(selectionAfterBlock).toBe("api-acct-fresh-healthy"); expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); }); + + test("keeps broker-sourced fresh Codex block when sibling selection sees healthy usage", async () => { + if (!authStorage || !store?.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-broker-fresh-blocked", "broker-fresh-blocked@example.com") }, + { type: "oauth", ...createCredential("acct-broker-fresh-healthy", "broker-fresh-healthy@example.com") }, + ]); + + usageByAccount.set( + "acct-broker-fresh-blocked", + createCodexUsageReport({ + accountId: "acct-broker-fresh-blocked", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "broker-fresh-blocked@example.com", + accountId: "acct-broker-fresh-blocked", + }, + }), + ); + usageByAccount.set( + "acct-broker-fresh-healthy", + createCodexUsageReport({ + accountId: "acct-broker-fresh-healthy", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "broker-fresh-healthy@example.com", + accountId: "acct-broker-fresh-healthy", + }, + }), + ); + + const token = "codex-broker-fresh-block"; + const handle = startAuthBroker({ + storage: authStorage, + bind: "127.0.0.1:0", + bearerTokens: [token], + disableRefresher: true, + }); + try { + const clientA = new AuthBrokerClient({ url: handle.url, token }); + const clientB = new AuthBrokerClient({ url: handle.url, token }); + const initialResult = await clientB.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("expected initial broker snapshot"); + const blockedRow = initialResult.snapshot.credentials.find(entry => { + const credential = entry.credential; + return credential.type === "oauth" && credential.accountId === "acct-broker-fresh-blocked"; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + + const remoteStoreA = new RemoteAuthCredentialStore({ + client: clientA, + initialSnapshot: initialResult.snapshot, + streamSnapshots: false, + }); + const remoteStoreB = new RemoteAuthCredentialStore({ + client: clientB, + initialSnapshot: initialResult.snapshot, + streamSnapshots: false, + }); + const clientStorageA = new AuthStorage(remoteStoreA); + const clientStorageB = new AuthStorage(remoteStoreB); + await clientStorageA.reload(); + await clientStorageB.reload(); + try { + let blockedSessionId: string | undefined; + for (let index = 0; index < 100; index += 1) { + const sessionId = `codex-broker-fresh-block-selected-${index}`; + if ((await clientStorageA.getApiKey("openai-codex", sessionId)) === "api-acct-broker-fresh-blocked") { + blockedSessionId = sessionId; + break; + } + } + if (!blockedSessionId) throw new Error("expected client A to select the soon-blocked account"); + + const markResult = await clientStorageA.markUsageLimitReached("openai-codex", blockedSessionId, { + retryAfterMs: 6 * 24 * HOUR_MS, + }); + expect(markResult.switched).toBe(true); + + const updatedSnapshot = await clientB.fetchSnapshot({ + ifGenerationGt: initialResult.generation, + waitMs: 1000, + }); + if (updatedSnapshot.status !== 200) throw new Error("expected broker snapshot containing fresh block"); + + await remoteStoreB.refreshSnapshot(); + expect(remoteStoreB.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + + expect(await clientStorageB.getApiKey("openai-codex", "codex-broker-fresh-block-sibling")).toBe( + "api-acct-broker-fresh-healthy", + ); + expect(remoteStoreB.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + } finally { + clientStorageA.close(); + clientStorageB.close(); + remoteStoreA.close(); + remoteStoreB.close(); + } + } finally { + await handle.close(); + } + }); + test("an older in-flight healthy Codex usage report does not clear a newer usage-limit block", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); @@ -757,6 +876,7 @@ describe("AuthStorage codex oauth ranking", () => { blockScope: "shared", blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, }); + store.cleanExpiredCredentialBlocks?.(Date.now() + STALE_BLOCK_GUARD_MS); const token = "codex-broker-reconcile"; const handle = startAuthBroker({ From 497d385ce040574c3568505169b617570b847232 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 22:12:17 +0000 Subject: [PATCH 025/205] fix(auth): decoupled login success from model refresh Switched interactive OAuth login to start model discovery in the background after credentials are saved. Added a regression test that keeps model refresh pending and asserts the success transcript appears immediately. Fixes #4989 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../modes/controllers/selector-controller.ts | 2 +- .../selector-controller-login.test.ts | 65 +++++++++++++++++++ 3 files changed, 70 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/test/modes/controllers/selector-controller-login.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..68c8f2f09 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed interactive OAuth login success messages waiting on model discovery; `/login xai-oauth` now reports saved credentials immediately while model metadata refreshes in the background. ([#4989](https://github.com/can1357/oh-my-pi/issues/4989)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index 8e06c5887..15842c765 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -1179,7 +1179,7 @@ export class SelectorController { }, onManualCodeInput: useManualInput ? () => manualInput.waitForInput(providerId) : undefined, }); - await this.ctx.session.modelRegistry.refresh(); + this.ctx.session.modelRegistry.refreshInBackground(); const block = new TranscriptBlock(); block.addChild( new Text(theme.fg("success", `${theme.status.success} Successfully logged in to ${providerId}`), 1, 0), diff --git a/packages/coding-agent/test/modes/controllers/selector-controller-login.test.ts b/packages/coding-agent/test/modes/controllers/selector-controller-login.test.ts new file mode 100644 index 000000000..c062c86db --- /dev/null +++ b/packages/coding-agent/test/modes/controllers/selector-controller-login.test.ts @@ -0,0 +1,65 @@ +import { beforeAll, describe, expect, it, vi } from "bun:test"; +import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; +import type { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; + +interface RenderableBlock { + render(width: number): string[]; +} + +function renderPresented(blocks: unknown[]): string { + return blocks + .flatMap(block => { + const maybeRenderable = block as Partial; + return maybeRenderable.render ? maybeRenderable.render(120) : [String(block)]; + }) + .join("\n"); +} + +beforeAll(async () => { + await initTheme(); +}); + +describe("SelectorController login", () => { + it("presents OAuth success as soon as credentials are saved", async () => { + const loginSaved = Promise.withResolvers(); + const presentedBlocks: unknown[] = []; + const authStorage = { + login: vi.fn(async () => { + loginSaved.resolve(); + }), + } as unknown as AuthStorage; + const refresh = vi.fn(() => new Promise(() => {})); + const refreshInBackground = vi.fn(); + const ctx = { + oauthManualInput: { + waitForInput: vi.fn(), + clear: vi.fn(), + }, + session: { + modelRegistry: { + authStorage, + refresh, + refreshInBackground, + }, + }, + showStatus: vi.fn(), + showError: vi.fn(), + present: vi.fn((block: unknown) => { + presentedBlocks.push(block); + }), + openInBrowser: vi.fn(), + } as unknown as InteractiveModeContext; + const controller = new SelectorController(ctx); + + void controller.showOAuthSelector("login", "xai-oauth"); + await loginSaved.promise; + await Promise.resolve(); + + expect(renderPresented(presentedBlocks)).toContain("Successfully logged in to xai-oauth"); + expect(refreshInBackground).toHaveBeenCalledTimes(1); + expect(refresh).not.toHaveBeenCalled(); + expect(ctx.showError).not.toHaveBeenCalled(); + }); +}); From d8745b62b417be9df135acf47c9706554faf1705 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 22:23:07 +0000 Subject: [PATCH 026/205] fix(ai): protected initial broker blocks - Protected Codex blocks present in RemoteAuthCredentialStore initial snapshots from immediate healthy-usage reconciliation. - Added regression coverage for clients that start after a broker peer already persisted a fresh 429 block. - Kept stale broker reconciliation explicit by expiring the test guard before the stale-block refresh path. Fixes #4980 --- packages/ai/src/auth-broker/remote-store.ts | 2 +- .../test/auth-storage-codex-selection.test.ts | 142 ++++++++++++++++++ 2 files changed, 143 insertions(+), 1 deletion(-) diff --git a/packages/ai/src/auth-broker/remote-store.ts b/packages/ai/src/auth-broker/remote-store.ts index a2c30eb61..c1552870d 100644 --- a/packages/ai/src/auth-broker/remote-store.ts +++ b/packages/ai/src/auth-broker/remote-store.ts @@ -225,7 +225,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { constructor(opts: RemoteAuthCredentialStoreOptions) { this.#client = opts.client; this.#streamSnapshots = opts.streamSnapshots ?? true; - this.#applySnapshot(opts.initialSnapshot ?? emptySnapshot(), opts.initialSnapshot?.generation ?? 0, false); + this.#applySnapshot(opts.initialSnapshot ?? emptySnapshot(), opts.initialSnapshot?.generation ?? 0); this.#onSnapshot = opts.onSnapshot; void this.#runBackground(); } diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 880074ad3..19f227eb3 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -737,6 +737,147 @@ describe("AuthStorage codex oauth ranking", () => { } }); + test("protects fresh Codex blocks present in the initial broker snapshot from healthy selection reconciliation", async () => { + if (!authStorage || !store?.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { + type: "oauth", + ...createCredential("acct-broker-initial-snapshot-blocked", "broker-initial-snapshot-blocked@example.com"), + }, + { + type: "oauth", + ...createCredential("acct-broker-initial-snapshot-healthy", "broker-initial-snapshot-healthy@example.com"), + }, + ]); + + usageByAccount.set( + "acct-broker-initial-snapshot-blocked", + createCodexUsageReport({ + accountId: "acct-broker-initial-snapshot-blocked", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "broker-initial-snapshot-blocked@example.com", + accountId: "acct-broker-initial-snapshot-blocked", + }, + }), + ); + usageByAccount.set( + "acct-broker-initial-snapshot-healthy", + createCodexUsageReport({ + accountId: "acct-broker-initial-snapshot-healthy", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "broker-initial-snapshot-healthy@example.com", + accountId: "acct-broker-initial-snapshot-healthy", + }, + }), + ); + + const token = "codex-broker-initial-snapshot-block"; + const handle = startAuthBroker({ + storage: authStorage, + bind: "127.0.0.1:0", + bearerTokens: [token], + disableRefresher: true, + }); + try { + const clientA = new AuthBrokerClient({ url: handle.url, token }); + const clientB = new AuthBrokerClient({ url: handle.url, token }); + const clientAInitial = await clientA.fetchSnapshot(); + if (clientAInitial.status !== 200) throw new Error("expected client A broker snapshot"); + + const remoteStoreA = new RemoteAuthCredentialStore({ + client: clientA, + initialSnapshot: clientAInitial.snapshot, + streamSnapshots: false, + }); + const clientStorageA = new AuthStorage(remoteStoreA); + await clientStorageA.reload(); + try { + let blockedSessionId: string | undefined; + let blockedAccountId: string | undefined; + for (let index = 0; index < 100; index += 1) { + const sessionId = `codex-broker-initial-snapshot-block-selected-${index}`; + const apiKey = await clientStorageA.getApiKey("openai-codex", sessionId); + if ( + apiKey === "api-acct-broker-initial-snapshot-blocked" || + apiKey === "api-acct-broker-initial-snapshot-healthy" + ) { + blockedSessionId = sessionId; + blockedAccountId = apiKey.replace(/^api-/, ""); + break; + } + } + if (!blockedSessionId || !blockedAccountId) { + throw new Error("expected client A to select a Codex account to block"); + } + const healthyAccountId = + blockedAccountId === "acct-broker-initial-snapshot-blocked" + ? "acct-broker-initial-snapshot-healthy" + : "acct-broker-initial-snapshot-blocked"; + + const markResult = await clientStorageA.markUsageLimitReached("openai-codex", blockedSessionId, { + retryAfterMs: 6 * 24 * HOUR_MS, + }); + expect(markResult.switched).toBe(true); + + const snapshotWithBlock = await clientB.fetchSnapshot({ + ifGenerationGt: clientAInitial.generation, + waitMs: 1000, + }); + if (snapshotWithBlock.status !== 200) throw new Error("expected broker snapshot containing initial block"); + + const blockedRow = snapshotWithBlock.snapshot.credentials.find(entry => { + const credential = entry.credential; + return credential.type === "oauth" && credential.accountId === blockedAccountId; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + expect( + blockedRow.blocks?.some( + block => block.providerKey === "openai-codex:oauth" && block.blockScope === "shared", + ), + ).toBe(true); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + + const remoteStoreB = new RemoteAuthCredentialStore({ + client: clientB, + initialSnapshot: snapshotWithBlock.snapshot, + streamSnapshots: false, + }); + const clientStorageB = new AuthStorage(remoteStoreB); + await clientStorageB.reload(); + try { + expect(remoteStoreB.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + + expect(await clientStorageB.getApiKey("openai-codex", "codex-broker-initial-snapshot-sibling")).toBe( + `api-${healthyAccountId}`, + ); + expect(remoteStoreB.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + } finally { + clientStorageB.close(); + remoteStoreB.close(); + } + } finally { + clientStorageA.close(); + remoteStoreA.close(); + } + } finally { + await handle.close(); + } + }); + test("an older in-flight healthy Codex usage report does not clear a newer usage-limit block", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); @@ -904,6 +1045,7 @@ describe("AuthStorage codex oauth ranking", () => { try { expect(remoteStore.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + remoteStore.cleanExpiredCredentialBlocks(Date.now() + STALE_BLOCK_GUARD_MS); await clientStorage.fetchUsageReports(); From 4911ce02e85e3b8456eddba7795ae0488f1c2e0e Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 22:50:33 +0000 Subject: [PATCH 027/205] fix(ai): persisted codex block guard - Derived SQLite credential-block reconciliation delays from persisted updated_at so fresh blocks survive process restarts and sibling local stores. - Added reopened-SQLite regression coverage for healthy usage lag after a persisted Codex 429 block. - Aged explicit stale-block fixtures by moving updated_at outside the guard window. Fixes #4980 --- packages/ai/src/auth-storage.ts | 21 +++- .../test/auth-storage-codex-selection.test.ts | 103 +++++++++++++++++- 2 files changed, 118 insertions(+), 6 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 36eb052d9..478d540f4 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -4941,6 +4941,7 @@ type CredentialBlockRow = { provider_key: string; block_scope: string; blocked_until_ms: number; + updated_at: number; }; type SerializedCredentialRecord = { @@ -5203,10 +5204,10 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { ); this.#deleteExpiredCacheStmt = this.#db.prepare(`DELETE FROM cache WHERE expires_at <= ${SQLITE_NOW_EPOCH}`); this.#getCredentialBlockStmt = this.#db.prepare( - "SELECT blocked_until_ms FROM auth_credential_blocks WHERE credential_id = ? AND provider_key = ? AND block_scope = ? AND blocked_until_ms > ?", + "SELECT blocked_until_ms, updated_at FROM auth_credential_blocks WHERE credential_id = ? AND provider_key = ? AND block_scope = ? AND blocked_until_ms > ?", ); this.#listCredentialBlocksByCredentialStmt = this.#db.prepare( - "SELECT credential_id, provider_key, block_scope, blocked_until_ms FROM auth_credential_blocks WHERE credential_id = ? AND blocked_until_ms > ? ORDER BY provider_key ASC, block_scope ASC", + "SELECT credential_id, provider_key, block_scope, blocked_until_ms, updated_at FROM auth_credential_blocks WHERE credential_id = ? AND blocked_until_ms > ? ORDER BY provider_key ASC, block_scope ASC", ); this.#upsertCredentialBlockStmt = this.#db.prepare( `INSERT INTO auth_credential_blocks (credential_id, provider_key, block_scope, blocked_until_ms, updated_at) @@ -5820,14 +5821,24 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { const nowMs = Date.now(); this.#deleteExpiredCredentialBlocksStmt.run(nowMs); const row = this.#getCredentialBlockStmt.get(credentialId, providerKey, blockScope, nowMs) as - | { blocked_until_ms?: number } + | { blocked_until_ms?: number; updated_at?: number } | undefined; return typeof row?.blocked_until_ms === "number" ? row.blocked_until_ms : undefined; } getCredentialBlockReconcileAfter(credentialId: number, providerKey: string, blockScope: string): number | undefined { - if (this.getCredentialBlock(credentialId, providerKey, blockScope) === undefined) return undefined; - return this.#credentialBlockReconcileAfter.get(`${credentialId}\0${providerKey}\0${blockScope}`); + const nowMs = Date.now(); + this.#deleteExpiredCredentialBlocksStmt.run(nowMs); + const row = this.#getCredentialBlockStmt.get(credentialId, providerKey, blockScope, nowMs) as + | { blocked_until_ms?: number; updated_at?: number } + | undefined; + if (typeof row?.blocked_until_ms !== "number") return undefined; + const memoryReconcileAfter = + this.#credentialBlockReconcileAfter.get(`${credentialId}\0${providerKey}\0${blockScope}`) ?? 0; + const persistedReconcileAfter = + typeof row.updated_at === "number" ? row.updated_at * 1000 + USAGE_REPORT_TTL_MS : 0; + const reconcileAfter = Math.max(memoryReconcileAfter, persistedReconcileAfter); + return reconcileAfter > nowMs ? Math.min(row.blocked_until_ms, reconcileAfter) : undefined; } upsertCredentialBlock(block: StoredCredentialBlock): void { diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 19f227eb3..85c7c822c 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -1,3 +1,4 @@ +import { Database } from "bun:sqlite"; import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; @@ -15,6 +16,17 @@ const HOUR_MS = 60 * 60 * 1000; const FIVE_HOUR_MS = 5 * HOUR_MS; const STALE_BLOCK_GUARD_MS = 5 * 60_000 + 1; +function ageCredentialBlockRows(dbPath: string): void { + const db = new Database(dbPath); + try { + db.prepare("UPDATE auth_credential_blocks SET updated_at = ?").run( + Math.floor((Date.now() - STALE_BLOCK_GUARD_MS) / 1000), + ); + } finally { + db.close(); + } +} + type UsageWindowSpec = { usedFraction: number; resetInMs: number; @@ -145,6 +157,7 @@ function expectWeightedPreference(counts: Map, preferred: string describe("AuthStorage codex oauth ranking", () => { let tempDir = ""; let store: AuthCredentialStore | null = null; + let dbPath = ""; let authStorage: AuthStorage | null = null; const usageByAccount = new Map(); @@ -159,7 +172,8 @@ describe("AuthStorage codex oauth ranking", () => { beforeEach(async () => { tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-codex-selection-")); - store = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db")); + dbPath = path.join(tempDir, "agent.db"); + store = await SqliteAuthCredentialStore.open(dbPath); authStorage = new AuthStorage(store, { usageProviderResolver: provider => (provider === "openai-codex" ? usageProvider : undefined), }); @@ -354,6 +368,7 @@ describe("AuthStorage codex oauth ranking", () => { blockScope: "shared", blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, }); + ageCredentialBlockRows(dbPath); store.cleanExpiredCredentialBlocks?.(Date.now() + STALE_BLOCK_GUARD_MS); usageByAccount.set( @@ -426,6 +441,7 @@ describe("AuthStorage codex oauth ranking", () => { blockScope: "shared", blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, }); + ageCredentialBlockRows(dbPath); store.cleanExpiredCredentialBlocks?.(Date.now() + STALE_BLOCK_GUARD_MS); const fiveHourWindow: UsageWindowConfig = { @@ -622,6 +638,90 @@ describe("AuthStorage codex oauth ranking", () => { expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); }); + test("protects a fresh Codex block after reopening SQLite storage", async () => { + if (!authStorage || !store?.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-reopened-blocked", "reopened-blocked@example.com") }, + { type: "oauth", ...createCredential("acct-reopened-healthy", "reopened-healthy@example.com") }, + ]); + + usageByAccount.set( + "acct-reopened-blocked", + createCodexUsageReport({ + accountId: "acct-reopened-blocked", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "reopened-blocked@example.com", + accountId: "acct-reopened-blocked", + }, + }), + ); + usageByAccount.set( + "acct-reopened-healthy", + createCodexUsageReport({ + accountId: "acct-reopened-healthy", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "reopened-healthy@example.com", + accountId: "acct-reopened-healthy", + }, + }), + ); + + const firstSelectionSessionId = "codex-reopened-fresh-block-initial"; + const firstSelection = await authStorage.getApiKey("openai-codex", firstSelectionSessionId); + if (!firstSelection) throw new Error("expected initial Codex credential"); + + const blockedAccountId = firstSelection.replace(/^api-/, ""); + const healthyAccountId = + blockedAccountId === "acct-reopened-blocked" ? "acct-reopened-healthy" : "acct-reopened-blocked"; + + const blockedRow = store.listAuthCredentials("openai-codex").find(row => { + const credential = row.credential; + return credential.type === "oauth" && credential.accountId === blockedAccountId; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + + const markResult = await authStorage.markUsageLimitReached("openai-codex", firstSelectionSessionId, { + retryAfterMs: 6 * 24 * HOUR_MS, + }); + + expect(markResult.switched).toBe(true); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + + authStorage.close(); + authStorage = null; + store = null; + + const reopenedStore = await SqliteAuthCredentialStore.open(dbPath); + const reopenedAuthStorage = new AuthStorage(reopenedStore, { + usageProviderResolver: provider => (provider === "openai-codex" ? usageProvider : undefined), + }); + try { + await reopenedAuthStorage.reload(); + + const selectionAfterReopen = await reopenedAuthStorage.getApiKey( + "openai-codex", + "codex-reopened-fresh-block-sibling", + ); + expect(selectionAfterReopen).toBe(`api-${healthyAccountId}`); + expect(reopenedStore.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBeDefined(); + } finally { + reopenedAuthStorage.close(); + } + }); + test("keeps broker-sourced fresh Codex block when sibling selection sees healthy usage", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); @@ -1017,6 +1117,7 @@ describe("AuthStorage codex oauth ranking", () => { blockScope: "shared", blockedUntilMs: Date.now() + 6 * 24 * HOUR_MS, }); + ageCredentialBlockRows(dbPath); store.cleanExpiredCredentialBlocks?.(Date.now() + STALE_BLOCK_GUARD_MS); const token = "codex-broker-reconcile"; From 4f4f852ad6842da7c3410ff1276a6cfba8e6d1a0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 23:14:17 +0000 Subject: [PATCH 028/205] fix(coding-agent): restored ask tool timeout Ensured ask tool timeouts abort stalled UI selectors and return the recommended option instead of hanging. Added regression coverage for selectors that never settle. Fixes #4995 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/tools/ask.ts | 45 ++++++--- .../coding-agent/test/ask-timeout.test.ts | 91 +++++++++++++++++++ 3 files changed, 128 insertions(+), 12 deletions(-) create mode 100644 packages/coding-agent/test/ask-timeout.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..ff2122782 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the ask tool timeout so it auto-selects the recommended option even when the UI selector does not settle on its own. ([#4995](https://github.com/can1357/oh-my-pi/issues/4995)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index 42de7f930..c58ef975a 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -432,10 +432,16 @@ async function askSingleQuestion( const helpText = navigation ? "up/down navigate enter select ←/→ question esc cancel" : "up/down navigate enter select esc cancel"; + const timeoutController = typeof timeout === "number" && timeout > 0 ? new AbortController() : undefined; + const dialogSignal = + signal && timeoutController + ? AbortSignal.any([signal, timeoutController.signal]) + : (timeoutController?.signal ?? signal); + let timeoutId: NodeJS.Timeout | undefined; const dialogOptions = { initialIndex, timeout, - signal, + signal: dialogSignal, outline: true, onTimeout, helpText, @@ -454,18 +460,33 @@ async function askSingleQuestion( : undefined, }; const startMs = Date.now(); - const choice = signal - ? await untilAborted(signal, () => ui.select(prompt, optionsToShow, dialogOptions)) - : await ui.select(prompt, optionsToShow, dialogOptions); - if (!timeoutTriggered && choice === undefined && typeof timeout === "number") { - // Fallback for UI surfaces that enforce `timeout` without invoking - // `onTimeout`: their auto-cancel resolves right at the deadline. A - // cancel arriving well past the deadline is a deliberate user Esc on - // a surface that kept the dialog open — keep treating it as a cancel. - const elapsed = Date.now() - startMs; - timeoutTriggered = elapsed >= timeout && elapsed <= timeout + TIMEOUT_DETECTION_TOLERANCE_MS; + if (timeoutController && typeof timeout === "number") { + timeoutId = setTimeout(() => { + timeoutTriggered = true; + timeoutController.abort(); + }, timeout); + } + try { + const choice = dialogSignal + ? await untilAborted(dialogSignal, () => ui.select(prompt, optionsToShow, dialogOptions)) + : await ui.select(prompt, optionsToShow, dialogOptions); + if (!timeoutTriggered && choice === undefined && typeof timeout === "number") { + // Fallback for UI surfaces that enforce `timeout` without invoking + // `onTimeout`: their auto-cancel resolves right at the deadline. A + // cancel arriving well past the deadline is a deliberate user Esc on + // a surface that kept the dialog open — keep treating it as a cancel. + const elapsed = Date.now() - startMs; + timeoutTriggered = elapsed >= timeout && elapsed <= timeout + TIMEOUT_DETECTION_TOLERANCE_MS; + } + return { choice, timedOut: timeoutTriggered, navigation: navigationAction }; + } catch (error) { + if (timeoutTriggered && error instanceof Error && error.name === "AbortError") { + return { choice: undefined, timedOut: true, navigation: navigationAction }; + } + throw error; + } finally { + clearTimeout(timeoutId); } - return { choice, timedOut: timeoutTriggered, navigation: navigationAction }; }; const promptForCustomInput = async ( diff --git a/packages/coding-agent/test/ask-timeout.test.ts b/packages/coding-agent/test/ask-timeout.test.ts new file mode 100644 index 000000000..ebbf14f67 --- /dev/null +++ b/packages/coding-agent/test/ask-timeout.test.ts @@ -0,0 +1,91 @@ +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import type { AgentToolContext, AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import { getThemeByName, setThemeInstance } from "../src/modes/theme/theme"; +import type { ToolSession } from "../src/tools"; +import { AskTool, type AskToolDetails } from "../src/tools/ask"; + +type AskExecutionResult = AgentToolResult; + +async function drainMicrotasks(): Promise { + await Promise.resolve(); + await Promise.resolve(); +} + +function createAskTool(): AskTool { + return new AskTool({ + hasUI: true, + settings: { + get(key: string): unknown { + if (key === "ask.timeout") return 0.01; + if (key === "ask.notify") return "off"; + if (key === "speech.enabled") return false; + return undefined; + }, + }, + getPlanModeState: () => ({ enabled: false }), + } as unknown as ToolSession); +} + +describe("AskTool timeout", () => { + beforeAll(async () => { + const loaded = await getThemeByName("dark"); + if (!loaded) throw new Error("theme unavailable"); + setThemeInstance(loaded); + }); + + afterEach(() => { + vi.useRealTimers(); + vi.restoreAllMocks(); + }); + + it("auto-selects the recommended option when the selector does not settle", async () => { + vi.useFakeTimers(); + const select = vi.fn(() => new Promise(() => {})); + const abort = vi.fn(); + const context = { + hasUI: true, + ui: { + select, + editor: vi.fn(), + }, + abort, + } as unknown as AgentToolContext; + let result: AskExecutionResult | undefined; + let rejection: unknown; + + void createAskTool() + .execute( + "ask-timeout", + { + questions: [ + { + id: "db", + question: "Which database?", + options: [{ label: "SQLite" }, { label: "Postgres" }], + recommended: 1, + }, + ], + }, + undefined, + undefined, + context, + ) + .then( + value => { + result = value; + }, + error => { + rejection = error; + }, + ); + + await drainMicrotasks(); + vi.advanceTimersByTime(10); + await drainMicrotasks(); + + expect(rejection).toBeUndefined(); + expect(result?.details?.selectedOptions).toEqual(["Postgres"]); + expect(result?.details?.timedOut).toBe(true); + expect(abort).not.toHaveBeenCalled(); + }); +}); From 6d7341b139f61f3bce4573eedab7a7396991e03a Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 23:18:36 +0000 Subject: [PATCH 029/205] fix(ai): refreshed broker block guards - Sent credential-block updatedAtMs through broker snapshots so same-deadline block refreshes are observable by remote clients. - Refreshed RemoteAuthCredentialStore reconciliation guards when updatedAtMs changes even if blockedUntilMs is unchanged. - Added broker regression coverage for same-deadline Codex block re-upserts after a local guard expires. Fixes #4980 --- packages/ai/src/auth-broker/remote-store.ts | 23 ++- packages/ai/src/auth-broker/server.ts | 1 + packages/ai/src/auth-broker/wire-schemas.ts | 1 + packages/ai/src/auth-storage.ts | 3 + .../test/auth-storage-codex-selection.test.ts | 144 ++++++++++++++++++ 5 files changed, 166 insertions(+), 6 deletions(-) diff --git a/packages/ai/src/auth-broker/remote-store.ts b/packages/ai/src/auth-broker/remote-store.ts index c1552870d..4351fe616 100644 --- a/packages/ai/src/auth-broker/remote-store.ts +++ b/packages/ai/src/auth-broker/remote-store.ts @@ -51,7 +51,9 @@ function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: Credenti if (provider !== 0) return provider; const scope = a.blockScope.localeCompare(b.blockScope); if (scope !== 0) return scope; - return a.blockedUntilMs - b.blockedUntilMs; + const blockedUntil = a.blockedUntilMs - b.blockedUntilMs; + if (blockedUntil !== 0) return blockedUntil; + return (a.updatedAtMs ?? 0) - (b.updatedAtMs ?? 0); } function toCredentialBlockSnapshot(block: StoredCredentialBlock): CredentialBlockSnapshot { @@ -59,6 +61,7 @@ function toCredentialBlockSnapshot(block: StoredCredentialBlock): CredentialBloc providerKey: block.providerKey, blockScope: block.blockScope, blockedUntilMs: block.blockedUntilMs, + ...(block.updatedAtMs !== undefined ? { updatedAtMs: block.updatedAtMs } : {}), }; } @@ -75,7 +78,8 @@ function credentialBlockSnapshotsEqual( if ( leftBlock.providerKey !== rightBlock.providerKey || leftBlock.blockScope !== rightBlock.blockScope || - leftBlock.blockedUntilMs !== rightBlock.blockedUntilMs + leftBlock.blockedUntilMs !== rightBlock.blockedUntilMs || + leftBlock.updatedAtMs !== rightBlock.updatedAtMs ) { return false; } @@ -256,10 +260,13 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { } } #protectNewSnapshotBlocks(previous: readonly SnapshotEntry[], next: readonly SnapshotEntry[], nowMs: number): void { - const previousBlocksByKey = new Map(); + const previousBlocksByKey = new Map(); for (const entry of previous) { for (const block of entry.blocks ?? []) { - previousBlocksByKey.set(`${entry.id}\0${block.providerKey}\0${block.blockScope}`, block.blockedUntilMs); + previousBlocksByKey.set( + `${entry.id}\0${block.providerKey}\0${block.blockScope}`, + `${block.blockedUntilMs}\0${block.updatedAtMs ?? ""}`, + ); } } const activeKeys = new Set(); @@ -267,10 +274,12 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { for (const block of entry.blocks ?? []) { const key = `${entry.id}\0${block.providerKey}\0${block.blockScope}`; activeKeys.add(key); - if (previousBlocksByKey.get(key) === block.blockedUntilMs) continue; + const signature = `${block.blockedUntilMs}\0${block.updatedAtMs ?? ""}`; + if (previousBlocksByKey.get(key) === signature) continue; + const updatedAtMs = block.updatedAtMs ?? nowMs; this.#credentialBlockReconcileAfter.set( key, - Math.min(block.blockedUntilMs, nowMs + CREDENTIAL_BLOCK_RECONCILE_DELAY_MS), + Math.min(block.blockedUntilMs, updatedAtMs + CREDENTIAL_BLOCK_RECONCILE_DELAY_MS), ); } } @@ -440,6 +449,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { providerKey: block.providerKey, blockScope: block.blockScope, blockedUntilMs: block.blockedUntilMs, + updatedAtMs: block.updatedAtMs, }); } } @@ -691,6 +701,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { providerKey: block.providerKey, blockScope: block.blockScope, blockedUntilMs: block.blockedUntilMs, + ...(block.updatedAtMs !== undefined ? { updatedAtMs: block.updatedAtMs } : {}), })) .sort(compareCredentialBlockSnapshots); if (blocks.length > 0) return { ...entry, blocks }; diff --git a/packages/ai/src/auth-broker/server.ts b/packages/ai/src/auth-broker/server.ts index 820ea8bd6..bf5b8afb3 100644 --- a/packages/ai/src/auth-broker/server.ts +++ b/packages/ai/src/auth-broker/server.ts @@ -290,6 +290,7 @@ function buildCredentialBlockGroups( providerKey: block.providerKey, blockScope: block.blockScope, blockedUntilMs: block.blockedUntilMs, + updatedAtMs: block.updatedAtMs, }; const existing = byCredentialId.get(block.credentialId); if (existing) { diff --git a/packages/ai/src/auth-broker/wire-schemas.ts b/packages/ai/src/auth-broker/wire-schemas.ts index 741a9ea11..35180ebf1 100644 --- a/packages/ai/src/auth-broker/wire-schemas.ts +++ b/packages/ai/src/auth-broker/wire-schemas.ts @@ -74,6 +74,7 @@ export const credentialBlockSnapshotSchema = type({ providerKey: type("string").atLeastLength(1), blockScope: "string", blockedUntilMs: "number", + "updatedAtMs?": "number", }); export const snapshotEntrySchema = type({ diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 478d540f4..c0f2b6eda 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -134,6 +134,8 @@ export interface StoredCredentialBlock { blockScope: string; /** Epoch milliseconds. */ blockedUntilMs: number; + /** Last row update timestamp in epoch milliseconds, when provided by the backing store. */ + updatedAtMs?: number; } /** @@ -5884,6 +5886,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { providerKey: row.provider_key, blockScope: row.block_scope, blockedUntilMs: row.blocked_until_ms, + updatedAtMs: row.updated_at * 1000, }); } } diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 85c7c822c..d66c7afe7 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -837,6 +837,150 @@ describe("AuthStorage codex oauth ranking", () => { } }); + test("refreshes broker-sourced Codex block protection when the same deadline is re-upserted", async () => { + if (!authStorage || !store?.getCredentialBlock) { + throw new Error("test setup failed"); + } + + await authStorage.set("openai-codex", [ + { + type: "oauth", + ...createCredential("acct-broker-same-deadline-blocked", "broker-same-deadline-blocked@example.com"), + }, + { + type: "oauth", + ...createCredential("acct-broker-same-deadline-healthy", "broker-same-deadline-healthy@example.com"), + }, + ]); + + usageByAccount.set( + "acct-broker-same-deadline-blocked", + createCodexUsageReport({ + accountId: "acct-broker-same-deadline-blocked", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "broker-same-deadline-blocked@example.com", + accountId: "acct-broker-same-deadline-blocked", + }, + }), + ); + usageByAccount.set( + "acct-broker-same-deadline-healthy", + createCodexUsageReport({ + accountId: "acct-broker-same-deadline-healthy", + primary: { usedFraction: 0.2, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.3, resetInMs: WEEK_MS }, + metadata: { + allowed: true, + limitReached: false, + planType: "pro", + email: "broker-same-deadline-healthy@example.com", + accountId: "acct-broker-same-deadline-healthy", + }, + }), + ); + + const token = "codex-broker-same-deadline-block"; + const handle = startAuthBroker({ + storage: authStorage, + bind: "127.0.0.1:0", + bearerTokens: [token], + disableRefresher: true, + }); + try { + const clientA = new AuthBrokerClient({ url: handle.url, token }); + const clientB = new AuthBrokerClient({ url: handle.url, token }); + const initialResult = await clientB.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("expected initial broker snapshot"); + const blockedRow = initialResult.snapshot.credentials.find(entry => { + const credential = entry.credential; + return credential.type === "oauth" && credential.accountId === "acct-broker-same-deadline-blocked"; + }); + if (!blockedRow) throw new Error("expected blocked credential row"); + + const blockedUntilMs = Date.now() + 6 * 24 * HOUR_MS; + await clientA.upsertCredentialBlock(blockedRow.id, { + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs, + }); + + const initialUpdatedAtSec = Math.floor(Date.now() / 1000) - 1; + const db = new Database(dbPath); + try { + const result = db + .prepare( + "UPDATE auth_credential_blocks SET updated_at = ? WHERE credential_id = ? AND provider_key = ? AND block_scope = ?", + ) + .run(initialUpdatedAtSec, blockedRow.id, "openai-codex:oauth", "shared") as { changes: number }; + if (result.changes !== 1) throw new Error("expected to age the broker block update timestamp"); + } finally { + db.close(); + } + + const snapshotWithBlock = await clientB.fetchSnapshot({ + ifGenerationGt: initialResult.generation, + waitMs: 1000, + }); + if (snapshotWithBlock.status !== 200) + throw new Error("expected broker snapshot containing same-deadline block"); + const initialSnapshotBlock = snapshotWithBlock.snapshot.credentials + .find(entry => entry.id === blockedRow.id) + ?.blocks?.find(block => block.providerKey === "openai-codex:oauth" && block.blockScope === "shared"); + expect(initialSnapshotBlock?.blockedUntilMs).toBe(blockedUntilMs); + expect(initialSnapshotBlock?.updatedAtMs).toBe(initialUpdatedAtSec * 1000); + + const remoteStoreB = new RemoteAuthCredentialStore({ + client: clientB, + initialSnapshot: snapshotWithBlock.snapshot, + streamSnapshots: false, + }); + const clientStorageB = new AuthStorage(remoteStoreB); + await clientStorageB.reload(); + try { + expect(remoteStoreB.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBe(blockedUntilMs); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBe(blockedUntilMs); + + remoteStoreB.cleanExpiredCredentialBlocks(Date.now() + STALE_BLOCK_GUARD_MS); + + await clientA.upsertCredentialBlock(blockedRow.id, { + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs, + }); + const refreshedSnapshot = await clientB.fetchSnapshot({ + ifGenerationGt: snapshotWithBlock.generation, + waitMs: 1000, + }); + if (refreshedSnapshot.status !== 200) { + throw new Error("expected broker snapshot containing refreshed same-deadline block"); + } + + await remoteStoreB.refreshSnapshot(); + const refreshedBlock = remoteStoreB.snapshot.credentials + .find(entry => entry.id === blockedRow.id) + ?.blocks?.find(block => block.providerKey === "openai-codex:oauth" && block.blockScope === "shared"); + expect(refreshedBlock?.blockedUntilMs).toBe(blockedUntilMs); + expect(refreshedBlock?.updatedAtMs).toBeGreaterThan(initialSnapshotBlock!.updatedAtMs!); + + expect(await clientStorageB.getApiKey("openai-codex", "codex-broker-same-deadline-sibling")).toBe( + "api-acct-broker-same-deadline-healthy", + ); + expect(remoteStoreB.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBe(blockedUntilMs); + expect(store.getCredentialBlock(blockedRow.id, "openai-codex:oauth", "shared")).toBe(blockedUntilMs); + } finally { + clientStorageB.close(); + remoteStoreB.close(); + } + } finally { + await handle.close(); + } + }); + test("protects fresh Codex blocks present in the initial broker snapshot from healthy selection reconciliation", async () => { if (!authStorage || !store?.getCredentialBlock) { throw new Error("test setup failed"); From facfb3c35aa6cf2dd5a33d74473b5d8d587e78bc Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 23:37:50 +0000 Subject: [PATCH 030/205] fix(coding-agent): synced ask fallback timeout resets Reset the ask tool fallback timeout whenever the interactive selector resets its UI countdown, preventing late keypresses from falling back to the original recommended option. Refs #4995 --- .../src/extensibility/extensions/types.ts | 2 + .../src/modes/components/hook-selector.ts | 9 +- .../controllers/extension-ui-controller.ts | 1 + packages/coding-agent/src/tools/ask.ts | 24 +++-- .../coding-agent/test/ask-timeout.test.ts | 89 ++++++++++++++++++- 5 files changed, 114 insertions(+), 11 deletions(-) diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index cfff0ed01..59b67e3b9 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -125,6 +125,8 @@ export interface ExtensionUIDialogOptions { timeout?: number; /** Invoked when the UI times out while waiting for a selection/input */ onTimeout?: () => void; + /** Invoked when user input resets a UI-managed timeout countdown */ + onTimeoutReset?: () => void; /** Initial cursor position for select dialogs (0-indexed) */ initialIndex?: number; /** Render an outlined list for select dialogs */ diff --git a/packages/coding-agent/src/modes/components/hook-selector.ts b/packages/coding-agent/src/modes/components/hook-selector.ts index 99ea1ea1e..f5e5570b3 100644 --- a/packages/coding-agent/src/modes/components/hook-selector.ts +++ b/packages/coding-agent/src/modes/components/hook-selector.ts @@ -61,6 +61,7 @@ export interface HookSelectorOptions { tui?: TUI; timeout?: number; onTimeout?: () => void; + onTimeoutReset?: () => void; initialIndex?: number; outline?: boolean; maxVisible?: number; @@ -178,6 +179,7 @@ export class HookSelectorComponent extends Container { #onLeftCallback: (() => void) | undefined; #onRightCallback: (() => void) | undefined; #onExternalEditorCallback: (() => void) | undefined; + #onTimeoutResetCallback: (() => void) | undefined; #slider: HookSelectorSlider | undefined; #sliderIndex: number = 0; #sliderComponent: Text | undefined; @@ -213,6 +215,7 @@ export class HookSelectorComponent extends Container { this.#onLeftCallback = opts?.onLeft; this.#onRightCallback = opts?.onRight; this.#onExternalEditorCallback = opts?.onExternalEditor; + this.#onTimeoutResetCallback = opts?.onTimeoutReset; if (opts?.slider && opts.slider.segments.length > 0) { this.#slider = opts.slider; this.#sliderIndex = Math.max(0, Math.min(opts.slider.index, opts.slider.segments.length - 1)); @@ -633,8 +636,10 @@ export class HookSelectorComponent extends Container { } handleInput(keyData: string): void { - // Reset countdown on any interaction - this.#countdown?.reset(); + if (this.#countdown) { + this.#countdown.reset(); + this.#onTimeoutResetCallback?.(); + } if (matchesSelectCancel(keyData)) { this.#onCancelCallback(); diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index 684642b12..ef27b3644 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -624,6 +624,7 @@ export class ExtensionUiController { initialIndex: dialogOptions?.initialIndex, timeout: dialogOptions?.timeout, onTimeout: dialogOptions?.onTimeout, + onTimeoutReset: dialogOptions?.onTimeoutReset, tui: this.ctx.ui, outline: dialogOptions?.outline, disabledIndices: dialogOptions?.disabledIndices, diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index c58ef975a..15fdc8994 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -389,6 +389,7 @@ interface UIContext { signal?: AbortSignal; outline?: boolean; onTimeout?: () => void; + onTimeoutReset?: () => void; onLeft?: () => void; onRight?: () => void; helpText?: string; @@ -432,18 +433,29 @@ async function askSingleQuestion( const helpText = navigation ? "up/down navigate enter select ←/→ question esc cancel" : "up/down navigate enter select esc cancel"; - const timeoutController = typeof timeout === "number" && timeout > 0 ? new AbortController() : undefined; + const timeoutMs = typeof timeout === "number" && timeout > 0 ? timeout : undefined; + const timeoutController = timeoutMs === undefined ? undefined : new AbortController(); const dialogSignal = signal && timeoutController ? AbortSignal.any([signal, timeoutController.signal]) : (timeoutController?.signal ?? signal); let timeoutId: NodeJS.Timeout | undefined; + let timeoutStartedMs = Date.now(); + const armFallbackTimeout = (durationMs: number) => { + clearTimeout(timeoutId); + timeoutStartedMs = Date.now(); + timeoutId = setTimeout(() => { + timeoutTriggered = true; + timeoutController?.abort(); + }, durationMs); + }; const dialogOptions = { initialIndex, timeout, signal: dialogSignal, outline: true, onTimeout, + onTimeoutReset: timeoutMs === undefined ? undefined : () => armFallbackTimeout(timeoutMs), helpText, selectionMarker: marker?.selectionMarker, checkedIndices: marker?.checkedIndices, @@ -459,12 +471,8 @@ async function askSingleQuestion( } : undefined, }; - const startMs = Date.now(); - if (timeoutController && typeof timeout === "number") { - timeoutId = setTimeout(() => { - timeoutTriggered = true; - timeoutController.abort(); - }, timeout); + if (timeoutMs !== undefined) { + armFallbackTimeout(timeoutMs); } try { const choice = dialogSignal @@ -475,7 +483,7 @@ async function askSingleQuestion( // `onTimeout`: their auto-cancel resolves right at the deadline. A // cancel arriving well past the deadline is a deliberate user Esc on // a surface that kept the dialog open — keep treating it as a cancel. - const elapsed = Date.now() - startMs; + const elapsed = Date.now() - timeoutStartedMs; timeoutTriggered = elapsed >= timeout && elapsed <= timeout + TIMEOUT_DETECTION_TOLERANCE_MS; } return { choice, timedOut: timeoutTriggered, navigation: navigationAction }; diff --git a/packages/coding-agent/test/ask-timeout.test.ts b/packages/coding-agent/test/ask-timeout.test.ts index ebbf14f67..34a153207 100644 --- a/packages/coding-agent/test/ask-timeout.test.ts +++ b/packages/coding-agent/test/ask-timeout.test.ts @@ -1,10 +1,18 @@ import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import type { AgentToolContext, AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import type { TUI } from "@oh-my-pi/pi-tui"; +import type { ExtensionUIDialogOptions, ExtensionUISelectItem } from "../src/extensibility/extensions"; +import { HookSelectorComponent } from "../src/modes/components/hook-selector"; import { getThemeByName, setThemeInstance } from "../src/modes/theme/theme"; import type { ToolSession } from "../src/tools"; import { AskTool, type AskToolDetails } from "../src/tools/ask"; type AskExecutionResult = AgentToolResult; +type AskSelect = ( + title: string, + options: ExtensionUISelectItem[], + dialogOptions?: ExtensionUIDialogOptions, +) => Promise; async function drainMicrotasks(): Promise { await Promise.resolve(); @@ -40,7 +48,7 @@ describe("AskTool timeout", () => { it("auto-selects the recommended option when the selector does not settle", async () => { vi.useFakeTimers(); - const select = vi.fn(() => new Promise(() => {})); + const select = vi.fn(() => new Promise(() => {})); const abort = vi.fn(); const context = { hasUI: true, @@ -88,4 +96,83 @@ describe("AskTool timeout", () => { expect(result?.details?.timedOut).toBe(true); expect(abort).not.toHaveBeenCalled(); }); + + it("honors selector timeout resets before using the fallback timeout", async () => { + vi.useFakeTimers(); + let resetTimeout: (() => void) | undefined; + const select = vi.fn((_title, _options, dialogOptions) => { + resetTimeout = dialogOptions?.onTimeoutReset; + return new Promise(() => {}); + }); + const abort = vi.fn(); + const context = { + hasUI: true, + ui: { + select, + editor: vi.fn(), + }, + abort, + } as unknown as AgentToolContext; + let result: AskExecutionResult | undefined; + let rejection: unknown; + + void createAskTool() + .execute( + "ask-timeout", + { + questions: [ + { + id: "db", + question: "Which database?", + options: [{ label: "SQLite" }, { label: "Postgres" }], + recommended: 1, + }, + ], + }, + undefined, + undefined, + context, + ) + .then( + value => { + result = value; + }, + error => { + rejection = error; + }, + ); + + await drainMicrotasks(); + expect(resetTimeout).toBeDefined(); + + vi.advanceTimersByTime(9); + resetTimeout?.(); + vi.advanceTimersByTime(9); + await drainMicrotasks(); + + expect(result).toBeUndefined(); + + vi.advanceTimersByTime(1); + await drainMicrotasks(); + + expect(rejection).toBeUndefined(); + expect(result?.details?.selectedOptions).toEqual(["Postgres"]); + expect(result?.details?.timedOut).toBe(true); + expect(abort).not.toHaveBeenCalled(); + }); + + it("notifies callers when selector input resets the UI countdown", () => { + vi.useFakeTimers(); + const onTimeoutReset = vi.fn(); + const selector = new HookSelectorComponent("Pick one", ["SQLite", "Postgres"], vi.fn(), vi.fn(), { + timeout: 10, + tui: { requestRender: vi.fn() } as unknown as TUI, + onTimeoutReset, + }); + + selector.handleInput("j"); + + expect(onTimeoutReset).toHaveBeenCalledTimes(1); + selector.dispose(); + }); }); From 3af6088caa3f743c2aaacbd421d6bcf54163678c Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 23:57:41 +0000 Subject: [PATCH 031/205] fix(ai): suppressed xai reasoning summaries - Suppressed `reasoning.summary` for xai-oauth Responses requests while preserving other Responses providers' summary defaults. - Added regression coverage for the `xai-oauth/grok-4.5` reasoning payload and regenerated the bundled catalog entry for its effort dial. Fixes #4998 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/openai-responses.ts | 2 +- .../ai/test/xai-oauth-effort-strip.test.ts | 18 ++++ packages/catalog/src/models.json | 93 +++++++++++-------- 4 files changed, 76 insertions(+), 41 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 03dd313da..2fa23c5c1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `xai-oauth/grok-4.5` Responses requests to omit unsupported `reasoning.summary` while preserving the documented `reasoning.effort` payload ([#4998](https://github.com/can1357/oh-my-pi/issues/4998)). + ## [16.3.15] - 2026-07-09 ### Breaking Changes diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 3aa740236..8041c39e0 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -898,7 +898,7 @@ export function buildParams( omitReasoningEffort: options?.omitReasoningEffort, }); applyResponsesCompatPolicy(params, reasoningPolicy, { - reasoningSummary: options?.reasoningSummary, + reasoningSummary: model.provider === "xai-oauth" ? null : options?.reasoningSummary, mapEffort: effort => model.compat.reasoningEffortMap?.[effort as NonNullable] ?? model.thinking?.effortMap?.[effort as NonNullable] ?? diff --git a/packages/ai/test/xai-oauth-effort-strip.test.ts b/packages/ai/test/xai-oauth-effort-strip.test.ts index c513f380f..38f0b2312 100644 --- a/packages/ai/test/xai-oauth-effort-strip.test.ts +++ b/packages/ai/test/xai-oauth-effort-strip.test.ts @@ -1,4 +1,7 @@ import { describe, expect, test } from "bun:test"; +import { buildParams } from "@oh-my-pi/pi-ai/providers/openai-responses"; +import type { Context } from "@oh-my-pi/pi-ai/types"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; @@ -41,3 +44,18 @@ describe("effort-dial-less reasoner encoding (regression)", () => { expect(claude.thinking).toBeDefined(); }); }); + +const singleUserContext: Context = { + messages: [{ role: "user", content: "hello", timestamp: 0 }], +}; + +describe("xAI OAuth Responses reasoning payload (regression)", () => { + test("xai-oauth/grok-4.5 omits unsupported reasoning summary", () => { + const grok45 = getBundledModel<"openai-responses">("xai-oauth", "grok-4.5"); + if (!grok45) throw new Error("xai-oauth/grok-4.5 must be in bundled models.json"); + + const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined); + + expect(params.reasoning).toEqual({ effort: "high" }); + }); +}); diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 0d7cc7d3d..c93738add 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -11689,10 +11689,10 @@ "image" ], "cost": { - "input": 2, - "output": 10, - "cacheRead": 0.2, - "cacheWrite": 2.5 + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -13686,7 +13686,7 @@ "cacheRead": 0.3, "cacheWrite": 3.75 }, - "contextWindow": 1000000, + "contextWindow": 200000, "maxTokens": 64000, "thinking": { "mode": "budget", @@ -60418,8 +60418,6 @@ }, "contextWindow": 1050000, "maxTokens": 128000, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -60437,7 +60435,9 @@ "high": "xhigh", "xhigh": "max" } - } + }, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro" }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", @@ -60496,8 +60496,6 @@ }, "contextWindow": 1050000, "maxTokens": 128000, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -60515,7 +60513,9 @@ "high": "xhigh", "xhigh": "max" } - } + }, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro" }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", @@ -60574,8 +60574,6 @@ }, "contextWindow": 1050000, "maxTokens": 128000, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -60593,7 +60591,9 @@ "high": "xhigh", "xhigh": "max" } - } + }, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro" }, "o1": { "id": "o1", @@ -61445,8 +61445,6 @@ "maxTokens": 128000, "preferWebsockets": true, "priority": 3, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -61464,7 +61462,9 @@ "high": "xhigh", "xhigh": "max" } - } + }, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro" }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", @@ -61537,8 +61537,6 @@ "maxTokens": 128000, "preferWebsockets": true, "priority": 1, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -61556,7 +61554,9 @@ "high": "xhigh", "xhigh": "max" } - } + }, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro" }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", @@ -61629,8 +61629,6 @@ "maxTokens": 128000, "preferWebsockets": true, "priority": 2, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -61648,7 +61646,9 @@ "high": "xhigh", "xhigh": "max" } - } + }, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro" } }, "opencode": { @@ -85623,6 +85623,14 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false + }, "thinking": { "mode": "effort", "efforts": [ @@ -85635,14 +85643,6 @@ "effortMap": { "minimal": "low" } - }, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false } }, "grok-4.3": { @@ -85664,6 +85664,14 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false + }, "thinking": { "mode": "effort", "efforts": [ @@ -85676,14 +85684,6 @@ "effortMap": { "minimal": "low" } - }, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false } }, "grok-4.5": { @@ -85711,7 +85711,20 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": true + "omitReasoningEffort": false + }, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } } }, "grok-build": { From f4e9b81ce61bbe156290c7aadd18a44d70597de3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 00:11:48 +0000 Subject: [PATCH 032/205] fix(ai): preserved xai default reasoning behavior - Limited xAI reasoning summary suppression to explicit reasoning requests so default calls do not gain `reasoning.effort`. - Added regression coverage proving `xai-oauth/grok-4.5` leaves `params.reasoning` unset when reasoning was not requested. Fixes #4998 --- packages/ai/src/providers/openai-responses.ts | 8 +++++++- packages/ai/test/xai-oauth-effort-strip.test.ts | 9 +++++++++ 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 8041c39e0..937cbe87a 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -897,8 +897,14 @@ export function buildParams( filterReasoningHistory: options?.filterReasoningHistory, omitReasoningEffort: options?.omitReasoningEffort, }); + const reasoningSummary = + model.provider === "xai-oauth" + ? options?.reasoning === undefined + ? undefined + : null + : options?.reasoningSummary; applyResponsesCompatPolicy(params, reasoningPolicy, { - reasoningSummary: model.provider === "xai-oauth" ? null : options?.reasoningSummary, + reasoningSummary, mapEffort: effort => model.compat.reasoningEffortMap?.[effort as NonNullable] ?? model.thinking?.effortMap?.[effort as NonNullable] ?? diff --git a/packages/ai/test/xai-oauth-effort-strip.test.ts b/packages/ai/test/xai-oauth-effort-strip.test.ts index 38f0b2312..c84012057 100644 --- a/packages/ai/test/xai-oauth-effort-strip.test.ts +++ b/packages/ai/test/xai-oauth-effort-strip.test.ts @@ -50,6 +50,15 @@ const singleUserContext: Context = { }; describe("xAI OAuth Responses reasoning payload (regression)", () => { + test("xai-oauth/grok-4.5 leaves reasoning unset when no reasoning was requested", () => { + const grok45 = getBundledModel<"openai-responses">("xai-oauth", "grok-4.5"); + if (!grok45) throw new Error("xai-oauth/grok-4.5 must be in bundled models.json"); + + const { params } = buildParams(grok45, singleUserContext, undefined, undefined); + + expect(params.reasoning).toBeUndefined(); + }); + test("xai-oauth/grok-4.5 omits unsupported reasoning summary", () => { const grok45 = getBundledModel<"openai-responses">("xai-oauth", "grok-4.5"); if (!grok45) throw new Error("xai-oauth/grok-4.5 must be in bundled models.json"); From 1bb97efec8f3703570674e560f45862ab4625b9d Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 01:16:06 +0000 Subject: [PATCH 033/205] fix(agent): committed yield before budget abort Persist yield tool-call arguments as soon as an assistant turn commits the yield call, before the soft request budget guard can abort the session. Add a regression covering a yielding turn that crosses the budget threshold without a tool result event. Fixes #5006 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/task/executor.ts | 202 +++++++++++------- .../src/task/subprocess-tool-registry.ts | 6 + packages/coding-agent/src/tools/yield.ts | 27 +++ .../test/task/executor-wall-clock.test.ts | 68 ++++++ 5 files changed, 234 insertions(+), 73 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..afe509414 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed subagent `yield` tool calls being discarded when the soft request budget hard-aborted the same assistant turn before the yield result event landed. ([#5006](https://github.com/can1357/oh-my-pi/issues/5006)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 64a05f140..8262682f4 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -254,14 +254,18 @@ function withAbortTimeout( return wrappedPromise; } +function isRecord(value: unknown): value is Record { + if (!value || typeof value !== "object") return false; + return !Array.isArray(value); +} + function getReportFindingKey(value: unknown): string | null { - if (!value || typeof value !== "object") return null; - const record = value as Record; - const title = typeof record.title === "string" ? record.title : null; - const filePath = typeof record.file_path === "string" ? record.file_path : null; - const lineStart = typeof record.line_start === "number" ? record.line_start : null; - const lineEnd = typeof record.line_end === "number" ? record.line_end : null; - const priority = typeof record.priority === "string" ? record.priority : null; + if (!isRecord(value)) return null; + const title = typeof value.title === "string" ? value.title : null; + const filePath = typeof value.file_path === "string" ? value.file_path : null; + const lineStart = typeof value.line_start === "number" ? value.line_start : null; + const lineEnd = typeof value.line_end === "number" ? value.line_end : null; + const priority = typeof value.priority === "string" ? value.priority : null; if (!title || !filePath || lineStart === null || lineEnd === null) { return null; } @@ -908,6 +912,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { const abortSignal = abortController.signal; let activeSession: AgentSession | null = null; let yieldCalled = false; + const extractedToolCallIndexes = new Map(); // Accumulate usage incrementally from message_end events (no memory for streaming events) const accumulatedUsage: Usage = { @@ -1062,17 +1067,17 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { }; const getMessageContent = (message: unknown): unknown => { - if (message && typeof message === "object" && "content" in message) { - return (message as { content?: unknown }).content; + if (!isRecord(message) || !("content" in message)) { + return undefined; } - return undefined; + return message.content; }; const getMessageUsage = (message: unknown): unknown => { - if (message && typeof message === "object" && "usage" in message) { - return (message as { usage?: unknown }).usage; + if (!isRecord(message) || !("usage" in message)) { + return undefined; } - return undefined; + return message.usage; }; const updateRecentOutputLines = () => { @@ -1136,6 +1141,51 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { }); }; + const recordExtractedToolData = (toolName: string, toolCallId: string, data: unknown): void => { + progress.extractedToolData = progress.extractedToolData || {}; + const existing = progress.extractedToolData[toolName] || []; + const callKey = toolCallId.length > 0 ? `${toolName}\0${toolCallId}` : undefined; + if (callKey !== undefined) { + const existingIndex = extractedToolCallIndexes.get(callKey); + if (existingIndex !== undefined && existingIndex < existing.length) { + existing[existingIndex] = data; + progress.extractedToolData[toolName] = existing; + return; + } + } + const findingKey = toolName === "report_finding" ? getReportFindingKey(data) : null; + if (findingKey) { + const existingIndex = existing.findIndex(item => getReportFindingKey(item) === findingKey); + if (existingIndex >= 0) { + existing[existingIndex] = data; + if (callKey !== undefined) extractedToolCallIndexes.set(callKey, existingIndex); + } else { + existing.push(data); + if (callKey !== undefined) extractedToolCallIndexes.set(callKey, existing.length - 1); + } + } else { + existing.push(data); + if (callKey !== undefined) extractedToolCallIndexes.set(callKey, existing.length - 1); + } + progress.extractedToolData[toolName] = existing; + if (toolName === "yield") { + yieldCalled = true; + } + }; + + const commitToolCallData = (toolName: string, toolCallId: string, eventArgs: Record): boolean => { + const handler = subprocessToolRegistry.getHandler(toolName); + if (!handler?.extractCallData) return false; + const data = handler.extractCallData({ + toolName, + toolCallId, + args: eventArgs, + }); + if (data === undefined) return false; + recordExtractedToolData(toolName, toolCallId, data); + return true; + }; + const processEvent = (event: AgentEvent) => { if (resolved) return; const now = Date.now(); @@ -1151,14 +1201,19 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { case "tool_execution_start": { progress.toolCount++; progress.currentTool = event.toolName; - progress.currentToolArgs = extractToolArgsPreview( - (event as { toolArgs?: Record }).toolArgs || event.args || {}, - ); + let startArgs: Record = {}; + if ("toolArgs" in event && isRecord(event.toolArgs)) { + startArgs = event.toolArgs; + } else if (isRecord(event.args)) { + startArgs = event.args; + } + progress.currentToolArgs = extractToolArgsPreview(startArgs); progress.currentToolStartMs = now; const intent = event.intent?.trim(); if (intent) { progress.lastIntent = intent; } + commitToolCallData(event.toolName, event.toolCallId, startArgs); // Reset any prior in-flight task snapshot so we don't show stale // nested progress when the agent enters a fresh `task` call. if (event.toolName === "task") { @@ -1191,7 +1246,8 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { // Check for registered subagent tool handler const handler = subprocessToolRegistry.getHandler(event.toolName); - const eventArgs = (event as { args?: Record }).args ?? {}; + const eventRecord: unknown = event; + const eventArgs = isRecord(eventRecord) && isRecord(eventRecord.args) ? eventRecord.args : {}; if (handler) { // Extract data using handler if (handler.extractData) { @@ -1203,23 +1259,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { isError: event.isError, }); if (data !== undefined) { - progress.extractedToolData = progress.extractedToolData || {}; - const existing = progress.extractedToolData[event.toolName] || []; - const findingKey = event.toolName === "report_finding" ? getReportFindingKey(data) : null; - if (findingKey) { - const existingIndex = existing.findIndex(item => getReportFindingKey(item) === findingKey); - if (existingIndex >= 0) { - existing[existingIndex] = data; - } else { - existing.push(data); - } - } else { - existing.push(data); - } - progress.extractedToolData[event.toolName] = existing; - if (event.toolName === "yield") { - yieldCalled = true; - } + recordExtractedToolData(event.toolName, event.toolCallId, data); } } @@ -1284,7 +1324,24 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { const role = event.message?.role; if (role === "assistant") { progress.requests += 1; - if (softRequestBudget > 0 && !abortSent) { + const eventContent = isRecord(event) && "content" in event ? event.content : undefined; + const messageContent = getMessageContent(event.message) || eventContent; + if (messageContent && Array.isArray(messageContent)) { + for (const block of messageContent) { + if (!isRecord(block)) continue; + if (block.type === "text" && typeof block.text === "string") { + outputChunks.push(block.text); + continue; + } + if (block.type !== "toolCall" || typeof block.name !== "string") continue; + const toolCallId = typeof block.id === "string" ? block.id : ""; + const toolArgs = isRecord(block.arguments) ? block.arguments : {}; + if (commitToolCallData(block.name, toolCallId, toolArgs)) { + flushProgress = true; + } + } + } + if (softRequestBudget > 0 && !abortSent && !yieldCalled) { if (progress.requests >= softRequestBudget * 1.5) { requestAbort("budget"); } else if (softRequestBudgetNotice && !budgetSteerSent && progress.requests >= softRequestBudget) { @@ -1302,32 +1359,21 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { } } } - if (role === "assistant") { - const messageContent = - getMessageContent(event.message) || (event as AgentEvent & { content?: unknown }).content; - if (messageContent && Array.isArray(messageContent)) { - for (const block of messageContent) { - if (block.type === "text" && block.text) { - outputChunks.push(block.text); - } - } - } - } // Extract and accumulate usage (prefer message.usage, fallback to event.usage) - const messageUsage = getMessageUsage(event.message) || (event as AgentEvent & { usage?: unknown }).usage; - if (messageUsage && typeof messageUsage === "object") { + const eventUsage = isRecord(event) && "usage" in event ? event.usage : undefined; + const messageUsage = getMessageUsage(event.message) || eventUsage; + if (isRecord(messageUsage)) { // Only count assistant messages (not tool results, etc.) if (role === "assistant") { - const usageRecord = messageUsage as Record; - const costRecord = (messageUsage as { cost?: Record }).cost; + const costRecord = isRecord(messageUsage.cost) ? messageUsage.cost : undefined; hasUsage = true; - accumulatedUsage.input += getNumberField(usageRecord, "input") ?? 0; - accumulatedUsage.output += getNumberField(usageRecord, "output") ?? 0; - accumulatedUsage.cacheRead += getNumberField(usageRecord, "cacheRead") ?? 0; - accumulatedUsage.cacheWrite += getNumberField(usageRecord, "cacheWrite") ?? 0; - accumulatedUsage.totalTokens += getNumberField(usageRecord, "totalTokens") ?? 0; + accumulatedUsage.input += getNumberField(messageUsage, "input") ?? 0; + accumulatedUsage.output += getNumberField(messageUsage, "output") ?? 0; + accumulatedUsage.cacheRead += getNumberField(messageUsage, "cacheRead") ?? 0; + accumulatedUsage.cacheWrite += getNumberField(messageUsage, "cacheWrite") ?? 0; + accumulatedUsage.totalTokens += getNumberField(messageUsage, "totalTokens") ?? 0; accumulatedUsage.reasoningTokens = - (accumulatedUsage.reasoningTokens ?? 0) + (getNumberField(usageRecord, "reasoningTokens") ?? 0); + (accumulatedUsage.reasoningTokens ?? 0) + (getNumberField(messageUsage, "reasoningTokens") ?? 0); if (costRecord) { accumulatedUsage.cost.input += getNumberField(costRecord, "input") ?? 0; accumulatedUsage.cost.output += getNumberField(costRecord, "output") ?? 0; @@ -1342,7 +1388,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { // Track latest per-turn context size so the UI can show // "current context", not just cumulative billing volume. if (role === "assistant") { - const perTurnTotal = getNumberField(messageUsage as Record, "totalTokens"); + const perTurnTotal = getNumberField(messageUsage, "totalTokens"); if (perTurnTotal !== undefined && perTurnTotal > 0) { progress.contextTokens = perTurnTotal; } @@ -1587,35 +1633,45 @@ async function driveSessionToYield( } } - await awaitAbortable(session.waitForIdle()); + if (monitor.yieldCalled()) { + await session.waitForIdle(); + } else { + await awaitAbortable(session.waitForIdle()); + } const lastAssistant = session.getLastAssistantMessage(); if (lastAssistant) { if (lastAssistant.stopReason === "aborted") { - aborted = monitor.isAbortedRun(); - if (aborted) { - // A real caller signal or the wall-clock timer carries a precise - // reason (signal.reason / "runtime limit exceeded"). An internal - // turn abort does NOT — prefer the assistant message's own - // errorMessage ("Request was aborted" or a specific stream error) - // over the misleading "Cancelled by caller". - abortReasonText ??= monitor.hasExplicitAbortReason() - ? monitor.resolveAbortReasonText() - : lastAssistant.errorMessage?.trim() || monitor.resolveAbortReasonText(); + if (!monitor.yieldCalled() || monitor.runtimeLimitExceeded()) { + aborted = monitor.isAbortedRun(); + if (aborted) { + // A real caller signal or the wall-clock timer carries a precise + // reason (signal.reason / "runtime limit exceeded"). An internal + // turn abort does NOT — prefer the assistant message's own + // errorMessage ("Request was aborted" or a specific stream error) + // over the misleading "Cancelled by caller". + abortReasonText ??= monitor.hasExplicitAbortReason() + ? monitor.resolveAbortReasonText() + : lastAssistant.errorMessage?.trim() || monitor.resolveAbortReasonText(); + } + exitCode = 1; } - exitCode = 1; } else if (lastAssistant.stopReason === "error") { exitCode = 1; error ??= lastAssistant.errorMessage || "Subagent failed"; } } } catch (err) { - exitCode = 1; - if (!abortSignal.aborted) { - error = err instanceof Error ? err.stack || err.message : String(err); + if (abortSignal.aborted && monitor.yieldCalled() && !monitor.runtimeLimitExceeded()) { + exitCode = 0; + } else { + exitCode = 1; + if (!abortSignal.aborted) { + error = err instanceof Error ? err.stack || err.message : String(err); + } } } finally { - if (abortSignal.aborted) { + if (abortSignal.aborted && (!monitor.yieldCalled() || monitor.runtimeLimitExceeded())) { aborted = monitor.isAbortedRun(); if (aborted) { abortReasonText ??= monitor.resolveAbortReasonText(); diff --git a/packages/coding-agent/src/task/subprocess-tool-registry.ts b/packages/coding-agent/src/task/subprocess-tool-registry.ts index 7dd45d321..ac1fcd3bd 100644 --- a/packages/coding-agent/src/task/subprocess-tool-registry.ts +++ b/packages/coding-agent/src/task/subprocess-tool-registry.ts @@ -29,6 +29,12 @@ export interface SubprocessToolHandler { */ extractData?: (event: SubprocessToolEvent) => TData | undefined; + /** + * Extract structured data from a committed tool call before execution result + * delivery. Used for terminal tools whose call arguments are the durable result. + */ + extractCallData?: (event: SubprocessToolEvent) => TData | undefined; + /** * Whether this tool's completion should terminate the subprocess. * Return true to send SIGTERM after the tool completes. diff --git a/packages/coding-agent/src/tools/yield.ts b/packages/coding-agent/src/tools/yield.ts index 184645d94..ced2bf6c3 100644 --- a/packages/coding-agent/src/tools/yield.ts +++ b/packages/coding-agent/src/tools/yield.ts @@ -425,6 +425,33 @@ export class YieldTool implements AgentTool { // Register subprocess tool handler for extraction + termination. subprocessToolRegistry.register("yield", { + extractCallData: event => { + const raw = event.args; + const rawResult = raw?.result; + if (!rawResult || typeof rawResult !== "object" || Array.isArray(rawResult)) return undefined; + const resultRecord = rawResult as Record; + const errorMessage = typeof resultRecord.error === "string" ? resultRecord.error : undefined; + const data = resultRecord.data; + let yieldType: string | string[] | undefined; + try { + yieldType = parseYieldType(raw?.type); + } catch { + return undefined; + } + if (errorMessage !== undefined && data !== undefined) return undefined; + if (errorMessage === undefined && data === undefined && yieldType === undefined) return undefined; + if (errorMessage === undefined && data === null) return undefined; + return { + data, + status: errorMessage === undefined ? "success" : "aborted", + error: errorMessage, + type: yieldType, + useLastTurn: + errorMessage === undefined && data === undefined && yieldType !== undefined && !("error" in resultRecord) + ? true + : undefined, + }; + }, extractData: event => { const details = event.result?.details; if (!details || typeof details !== "object") return undefined; diff --git a/packages/coding-agent/test/task/executor-wall-clock.test.ts b/packages/coding-agent/test/task/executor-wall-clock.test.ts index 1a5f6ec56..94ae8ba63 100644 --- a/packages/coding-agent/test/task/executor-wall-clock.test.ts +++ b/packages/coding-agent/test/task/executor-wall-clock.test.ts @@ -260,6 +260,74 @@ describe("runSubprocess wall clock (task.maxRuntimeMs)", () => { expect(result.extractedToolData?.yield).toBeDefined(); }); + it("commits a yield tool call before the soft request budget aborts the turn", async () => { + const settings = Settings.isolated({ "task.softRequestBudget": 1 }); + const firstAssistantMessage = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "finishing the task" }], + stopReason: "stop" as const, + }; + const yieldAssistantMessage = { + role: "assistant" as const, + content: [ + { + type: "toolCall" as const, + id: "tool-yield-budget", + name: "yield", + arguments: { result: { data: { finished: true } } }, + }, + ], + stopReason: "toolUse" as const, + }; + let listenerRef: ((event: AgentSessionEvent) => void) | undefined; + let waitForIdleCalls = 0; + let abortCount = 0; + const session: Partial = { + state: { messages: [] } as never, + agent: { state: { systemPrompt: ["test"] } } as never, + extensionRunner: undefined as never, + sessionManager: { appendSessionInit: () => {} } as never, + getActiveToolNames: () => ["read", "yield"], + setActiveToolsByName: async () => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + listenerRef = listener; + return () => {}; + }, + prompt: async () => true, + waitForIdle: async () => { + waitForIdleCalls += 1; + if (waitForIdleCalls !== 1) return; + listenerRef?.({ + type: "message_end", + message: firstAssistantMessage, + } as unknown as AgentSessionEvent); + listenerRef?.({ + type: "message_end", + message: yieldAssistantMessage, + } as unknown as AgentSessionEvent); + }, + getLastAssistantMessage: () => yieldAssistantMessage as never, + abort: async () => { + abortCount += 1; + }, + dispose: async () => {}, + }; + mockCreateAgentSession(session as AgentSession); + + const result = await runSubprocess({ + ...baseOptions, + id: "subagent-soft-budget-yield", + settings, + }); + + expect(abortCount).toBe(0); + expect(result.aborted).toBe(false); + expect(result.exitCode).toBe(0); + expect(result.requests).toBe(2); + expect(result.abortReason).toBeUndefined(); + expect(JSON.parse(result.output)).toEqual({ finished: true }); + }); + it("propagates per-turn context tokens onto the SingleResult", async () => { // Async task consumers (index.ts) copy `singleResult.contextTokens` and // `singleResult.contextWindow` onto AgentProgress. This test pins the From 0d606f2f3002d76238384072270c2cf0f0b48930 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 01:21:33 +0000 Subject: [PATCH 034/205] fix(cli): preserved mcp tools with tools filter Interactive sessions defer MCP discovery, so CLI --tools produced an initial built-in-only active set and later MCP refreshes respected that filtered set. Force-activate deferred MCP tools when MCP discovery mode is disabled, matching the blocking startup path while leaving discovery-mode selection intact. Fixes #5013 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/sdk.ts | 14 ++---- .../coding-agent/src/session/agent-session.ts | 12 +++--- .../test/sdk-mcp-instructions.test.ts | 43 ++++++++++++++++++- 4 files changed, 55 insertions(+), 18 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..0d8d21bfb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `--tools` filtering in interactive sessions disabling deferred MCP tools; MCP tools discovered from configured servers now stay active when the flag limits only built-in tools. ([#5013](https://github.com/can1357/oh-my-pi/issues/5013)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0d27eac07..fa87bf6c3 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1763,7 +1763,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Discovery flipped on mid-flight: route the explicit request // through discovery-aware activation so selection persists. await liveSession.activateDiscoveredMCPTools(activation.explicitlyRequestedMCPToolNames); - } else if (!discoveryEnabled) { + } else if (!discoveryEnabled && !activateAll) { await liveSession.setActiveToolsByName([ ...liveSession.getActiveToolNames(), ...activation.explicitlyRequestedMCPToolNames, @@ -3055,16 +3055,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} try { await session.refreshMCPTools( tools, - deferMCPDiscoveryForUI && !mcpDiscoveryEnabled && options.toolNames === undefined - ? { activateAll: true } - : undefined, + deferMCPDiscoveryForUI && !mcpDiscoveryEnabled ? { activateAll: true } : undefined, ); - if (deferMCPDiscoveryForUI && !mcpDiscoveryEnabled && explicitlyRequestedMCPToolNames.length > 0) { - await session.setActiveToolsByName([ - ...session.getActiveToolNames(), - ...explicitlyRequestedMCPToolNames, - ]); - } } catch (error) { logger.warn("MCP tool refresh failed", { error: error instanceof Error ? error.message : String(error), @@ -3106,7 +3098,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} startDeferredMCPDiscovery?.(session, { mcpDiscoveryEnabled, explicitlyRequestedMCPToolNames, - activateAllMCPTools: !mcpDiscoveryEnabled && options.toolNames === undefined, + activateAllMCPTools: !mcpDiscoveryEnabled, }); return { diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..31cca6cfd 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -6593,8 +6593,8 @@ export class AgentSession { * * @param mcpTools The new MCP tools to register. * @param options.activateAll When true, force-activates every newly registered MCP tool - * regardless of prior selection state. Used when an ACP client provisions MCP servers - * for a session where MCP discovery is disabled. + * regardless of prior selection state. Used when MCP discovery is disabled and tools + * arrive after initial session activation. */ async refreshMCPTools(mcpTools: CustomTool[], options?: { activateAll?: boolean }): Promise { const previousSelectedMCPToolNames = this.getSelectedMCPToolNames(); @@ -6639,10 +6639,10 @@ export class AgentSession { if (options?.activateAll) { // Force-activate every newly registered MCP tool. This path is used - // when an ACP client provisions MCP servers for a session where MCP - // discovery is disabled — without it, getSelectedMCPToolNames() - // returns only already-active tools (circular deadlock: tools can - // only become active if they're already active). + // when MCP discovery is disabled and tools arrive after initial + // activation — without it, getSelectedMCPToolNames() returns only + // already-active tools (circular deadlock: tools can only become + // active if they're already active). const newMcpNames = mcpTools.map(t => t.name); const nextActive = [...new Set([...this.#getActiveNonMCPToolNames(), ...newMcpNames])]; await this.#applyActiveToolsByName(nextActive, { previousSelectedMCPToolNames }); diff --git a/packages/coding-agent/test/sdk-mcp-instructions.test.ts b/packages/coding-agent/test/sdk-mcp-instructions.test.ts index 686351288..042afb217 100644 --- a/packages/coding-agent/test/sdk-mcp-instructions.test.ts +++ b/packages/coding-agent/test/sdk-mcp-instructions.test.ts @@ -9,7 +9,7 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils"; -import { SERVER_INSTRUCTIONS } from "./fixtures/instructions-mcp"; +import { SERVER_INSTRUCTIONS, TOOL_NAME } from "./fixtures/instructions-mcp"; // Contract: a deferred interactive (`hasUI`) session runs MCP discovery off the // first-paint path, so an MCP server's `instructions` are not available when the @@ -19,6 +19,7 @@ import { SERVER_INSTRUCTIONS } from "./fixtures/instructions-mcp"; // guard: a prior version gated instruction inclusion on `!deferMCPDiscoveryForUI`, // which dropped server instructions permanently for every UI session. const FIXTURE_PATH = path.join(import.meta.dir, "fixtures", "instructions-mcp.ts"); +const MCP_TOOL_NAME = `mcp__instr_${TOOL_NAME}`; describe("createAgentSession MCP server instructions (deferred UI)", () => { let registryDir: string; @@ -113,4 +114,44 @@ describe("createAgentSession MCP server instructions (deferred UI)", () => { await session.dispose(); } }, 20_000); + + it("keeps MCP tools active after deferred discovery when CLI tool filtering names only built-ins", async () => { + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: tempDir, + modelRegistry, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated({}), + model: getBundledModel("openai", "gpt-4o-mini"), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableLsp: false, + skipPythonPreflight: true, + enableMCP: true, + hasUI: true, + toolNames: ["read"], + }); + try { + expect(session.getActiveToolNames()).toContain("read"); + + // Deferred MCP discovery is fire-and-forget and exposes no promise or + // event; fake timers cannot drive the real subprocess handshake, so we + // poll the live active-tool state and exit as soon as the fixture tool + // appears. + const deadline = Date.now() + 12_000; + let activeToolNames = session.getActiveToolNames(); + while (!activeToolNames.includes(MCP_TOOL_NAME) && Date.now() < deadline) { + await Bun.sleep(50); + activeToolNames = session.getActiveToolNames(); + } + + expect(activeToolNames).toContain("read"); + expect(activeToolNames).toContain(MCP_TOOL_NAME); + } finally { + await session.dispose(); + } + }, 20_000); }); From 4f689c0b1b6ecbb63e509770afdcec6f4355729c Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 01:47:08 +0000 Subject: [PATCH 035/205] fix(agent): validated pending yield commits Keep assistant yield tool calls pending until YieldTool.execute returns a successful tool result. Prevent invalid pre-execution yield arguments from bypassing schema retry handling while still suppressing the soft budget abort during validation. Fixes #5006 --- packages/coding-agent/src/task/executor.ts | 47 ++---- .../src/task/subprocess-tool-registry.ts | 6 - packages/coding-agent/src/tools/yield.ts | 27 --- .../test/task/executor-wall-clock.test.ts | 155 +++++++++++++++++- 4 files changed, 166 insertions(+), 69 deletions(-) diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 8262682f4..9f3ad5a3b 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -912,7 +912,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { const abortSignal = abortController.signal; let activeSession: AgentSession | null = null; let yieldCalled = false; - const extractedToolCallIndexes = new Map(); + let yieldCallPending = false; // Accumulate usage incrementally from message_end events (no memory for streaming events) const accumulatedUsage: Usage = { @@ -1141,51 +1141,27 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { }); }; - const recordExtractedToolData = (toolName: string, toolCallId: string, data: unknown): void => { + const recordExtractedToolData = (toolName: string, data: unknown): void => { progress.extractedToolData = progress.extractedToolData || {}; const existing = progress.extractedToolData[toolName] || []; - const callKey = toolCallId.length > 0 ? `${toolName}\0${toolCallId}` : undefined; - if (callKey !== undefined) { - const existingIndex = extractedToolCallIndexes.get(callKey); - if (existingIndex !== undefined && existingIndex < existing.length) { - existing[existingIndex] = data; - progress.extractedToolData[toolName] = existing; - return; - } - } const findingKey = toolName === "report_finding" ? getReportFindingKey(data) : null; if (findingKey) { const existingIndex = existing.findIndex(item => getReportFindingKey(item) === findingKey); if (existingIndex >= 0) { existing[existingIndex] = data; - if (callKey !== undefined) extractedToolCallIndexes.set(callKey, existingIndex); } else { existing.push(data); - if (callKey !== undefined) extractedToolCallIndexes.set(callKey, existing.length - 1); } } else { existing.push(data); - if (callKey !== undefined) extractedToolCallIndexes.set(callKey, existing.length - 1); } progress.extractedToolData[toolName] = existing; if (toolName === "yield") { yieldCalled = true; + yieldCallPending = false; } }; - const commitToolCallData = (toolName: string, toolCallId: string, eventArgs: Record): boolean => { - const handler = subprocessToolRegistry.getHandler(toolName); - if (!handler?.extractCallData) return false; - const data = handler.extractCallData({ - toolName, - toolCallId, - args: eventArgs, - }); - if (data === undefined) return false; - recordExtractedToolData(toolName, toolCallId, data); - return true; - }; - const processEvent = (event: AgentEvent) => { if (resolved) return; const now = Date.now(); @@ -1213,7 +1189,9 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { if (intent) { progress.lastIntent = intent; } - commitToolCallData(event.toolName, event.toolCallId, startArgs); + if (event.toolName === "yield" && !yieldCalled) { + yieldCallPending = true; + } // Reset any prior in-flight task snapshot so we don't show stale // nested progress when the agent enters a fresh `task` call. if (event.toolName === "task") { @@ -1259,10 +1237,14 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { isError: event.isError, }); if (data !== undefined) { - recordExtractedToolData(event.toolName, event.toolCallId, data); + recordExtractedToolData(event.toolName, data); } } + if (event.toolName === "yield") { + yieldCallPending = false; + } + // Check if handler wants to terminate the session if ( handler.shouldTerminate?.({ @@ -1334,14 +1316,13 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { continue; } if (block.type !== "toolCall" || typeof block.name !== "string") continue; - const toolCallId = typeof block.id === "string" ? block.id : ""; - const toolArgs = isRecord(block.arguments) ? block.arguments : {}; - if (commitToolCallData(block.name, toolCallId, toolArgs)) { + if (block.name === "yield" && !yieldCalled) { + yieldCallPending = true; flushProgress = true; } } } - if (softRequestBudget > 0 && !abortSent && !yieldCalled) { + if (softRequestBudget > 0 && !abortSent && !yieldCalled && !yieldCallPending) { if (progress.requests >= softRequestBudget * 1.5) { requestAbort("budget"); } else if (softRequestBudgetNotice && !budgetSteerSent && progress.requests >= softRequestBudget) { diff --git a/packages/coding-agent/src/task/subprocess-tool-registry.ts b/packages/coding-agent/src/task/subprocess-tool-registry.ts index ac1fcd3bd..7dd45d321 100644 --- a/packages/coding-agent/src/task/subprocess-tool-registry.ts +++ b/packages/coding-agent/src/task/subprocess-tool-registry.ts @@ -29,12 +29,6 @@ export interface SubprocessToolHandler { */ extractData?: (event: SubprocessToolEvent) => TData | undefined; - /** - * Extract structured data from a committed tool call before execution result - * delivery. Used for terminal tools whose call arguments are the durable result. - */ - extractCallData?: (event: SubprocessToolEvent) => TData | undefined; - /** * Whether this tool's completion should terminate the subprocess. * Return true to send SIGTERM after the tool completes. diff --git a/packages/coding-agent/src/tools/yield.ts b/packages/coding-agent/src/tools/yield.ts index ced2bf6c3..184645d94 100644 --- a/packages/coding-agent/src/tools/yield.ts +++ b/packages/coding-agent/src/tools/yield.ts @@ -425,33 +425,6 @@ export class YieldTool implements AgentTool { // Register subprocess tool handler for extraction + termination. subprocessToolRegistry.register("yield", { - extractCallData: event => { - const raw = event.args; - const rawResult = raw?.result; - if (!rawResult || typeof rawResult !== "object" || Array.isArray(rawResult)) return undefined; - const resultRecord = rawResult as Record; - const errorMessage = typeof resultRecord.error === "string" ? resultRecord.error : undefined; - const data = resultRecord.data; - let yieldType: string | string[] | undefined; - try { - yieldType = parseYieldType(raw?.type); - } catch { - return undefined; - } - if (errorMessage !== undefined && data !== undefined) return undefined; - if (errorMessage === undefined && data === undefined && yieldType === undefined) return undefined; - if (errorMessage === undefined && data === null) return undefined; - return { - data, - status: errorMessage === undefined ? "success" : "aborted", - error: errorMessage, - type: yieldType, - useLastTurn: - errorMessage === undefined && data === undefined && yieldType !== undefined && !("error" in resultRecord) - ? true - : undefined, - }; - }, extractData: event => { const details = event.result?.details; if (!details || typeof details !== "object") return undefined; diff --git a/packages/coding-agent/test/task/executor-wall-clock.test.ts b/packages/coding-agent/test/task/executor-wall-clock.test.ts index 94ae8ba63..6e145fade 100644 --- a/packages/coding-agent/test/task/executor-wall-clock.test.ts +++ b/packages/coding-agent/test/task/executor-wall-clock.test.ts @@ -274,7 +274,7 @@ describe("runSubprocess wall clock (task.maxRuntimeMs)", () => { type: "toolCall" as const, id: "tool-yield-budget", name: "yield", - arguments: { result: { data: { finished: true } } }, + arguments: { result: { data: { finished: "unvalidated" } } }, }, ], stopReason: "toolUse" as const, @@ -282,6 +282,7 @@ describe("runSubprocess wall clock (task.maxRuntimeMs)", () => { let listenerRef: ((event: AgentSessionEvent) => void) | undefined; let waitForIdleCalls = 0; let abortCount = 0; + let abortCountBeforeYieldExecutionEnd: number | undefined; const session: Partial = { state: { messages: [] } as never, agent: { state: { systemPrompt: ["test"] } } as never, @@ -305,6 +306,17 @@ describe("runSubprocess wall clock (task.maxRuntimeMs)", () => { type: "message_end", message: yieldAssistantMessage, } as unknown as AgentSessionEvent); + abortCountBeforeYieldExecutionEnd = abortCount; + listenerRef?.({ + type: "tool_execution_end", + toolCallId: "tool-yield-budget", + toolName: "yield", + result: { + content: [{ type: "text", text: "Result submitted." }], + details: { status: "success", data: { finished: "validated" } }, + }, + isError: false, + } as AgentSessionEvent); }, getLastAssistantMessage: () => yieldAssistantMessage as never, abort: async () => { @@ -320,12 +332,149 @@ describe("runSubprocess wall clock (task.maxRuntimeMs)", () => { settings, }); - expect(abortCount).toBe(0); + expect(abortCountBeforeYieldExecutionEnd).toBe(0); expect(result.aborted).toBe(false); expect(result.exitCode).toBe(0); expect(result.requests).toBe(2); expect(result.abortReason).toBeUndefined(); - expect(JSON.parse(result.output)).toEqual({ finished: true }); + expect(JSON.parse(result.output)).toEqual({ finished: "validated" }); + }); + + it("does not finalize rejected yield arguments after crossing the soft request budget", async () => { + const settings = Settings.isolated({ "task.softRequestBudget": 1 }); + const firstAssistantMessage = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "finishing the task" }], + stopReason: "stop" as const, + }; + const rejectedYieldMessage = { + role: "assistant" as const, + content: [ + { + type: "toolCall" as const, + id: "tool-yield-rejected", + name: "yield", + arguments: { result: { data: { finished: "rejected-before-validation" } } }, + }, + ], + stopReason: "toolUse" as const, + }; + const validYieldMessage = { + role: "assistant" as const, + content: [ + { + type: "toolCall" as const, + id: "tool-yield-valid", + name: "yield", + arguments: { result: { data: { finished: "unvalidated-later" } } }, + }, + ], + stopReason: "toolUse" as const, + }; + let listenerRef: ((event: AgentSessionEvent) => void) | undefined; + let lastAssistantMessage: + | typeof firstAssistantMessage + | typeof rejectedYieldMessage + | typeof validYieldMessage + | undefined; + let waitForIdleCalls = 0; + let abortCount = 0; + let abortCountBeforeRejectedYieldExecutionEnd: number | undefined; + let abortCountBeforeValidYieldExecutionEnd: number | undefined; + const promptCalls: Array<{ text: string; options?: PromptOptions }> = []; + const session: Partial = { + state: { messages: [] } as never, + agent: { state: { systemPrompt: ["test"] } } as never, + extensionRunner: undefined as never, + sessionManager: { appendSessionInit: () => {} } as never, + getActiveToolNames: () => ["read", "yield"], + setActiveToolsByName: async () => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + listenerRef = listener; + return () => {}; + }, + prompt: async (text: string, options?: PromptOptions) => { + promptCalls.push({ text, options }); + return true; + }, + waitForIdle: async () => { + waitForIdleCalls += 1; + if (waitForIdleCalls === 1) { + lastAssistantMessage = firstAssistantMessage; + listenerRef?.({ + type: "message_end", + message: firstAssistantMessage, + } as unknown as AgentSessionEvent); + lastAssistantMessage = rejectedYieldMessage; + listenerRef?.({ + type: "message_end", + message: rejectedYieldMessage, + } as unknown as AgentSessionEvent); + abortCountBeforeRejectedYieldExecutionEnd = abortCount; + listenerRef?.({ + type: "tool_execution_end", + toolCallId: "tool-yield-rejected", + toolName: "yield", + result: { + content: [{ type: "text", text: "Yield rejected." }], + details: { status: "error", data: { finished: "rejected-before-validation" } }, + }, + isError: true, + } as AgentSessionEvent); + return; + } + if (waitForIdleCalls === 2) { + lastAssistantMessage = validYieldMessage; + listenerRef?.({ + type: "message_end", + message: validYieldMessage, + } as unknown as AgentSessionEvent); + abortCountBeforeValidYieldExecutionEnd = abortCount; + listenerRef?.({ + type: "tool_execution_end", + toolCallId: "tool-yield-valid", + toolName: "yield", + result: { + content: [{ type: "text", text: "Result submitted." }], + details: { status: "success", data: { finished: "validated-later" } }, + }, + isError: false, + } as AgentSessionEvent); + } + }, + getLastAssistantMessage: () => lastAssistantMessage as never, + abort: async () => { + abortCount += 1; + }, + dispose: async () => {}, + }; + mockCreateAgentSession(session as AgentSession); + + const result = await runSubprocess({ + ...baseOptions, + id: "subagent-soft-budget-rejected-yield", + settings, + }); + + expect(abortCountBeforeRejectedYieldExecutionEnd).toBe(0); + expect(abortCountBeforeValidYieldExecutionEnd).toBe(0); + expect(promptCalls.length).toBeGreaterThanOrEqual(2); + expect(promptCalls[1]?.options?.synthetic).toBe(true); + expect(result.aborted).toBe(false); + expect(result.exitCode).toBe(0); + expect(result.requests).toBe(3); + expect(result.abortReason).toBeUndefined(); + expect(JSON.parse(result.output)).toEqual({ finished: "validated-later" }); + expect(result.extractedToolData?.yield).toEqual([ + { + data: { finished: "validated-later" }, + status: "success", + error: undefined, + type: undefined, + useLastTurn: undefined, + schemaOverridden: undefined, + }, + ]); }); it("propagates per-turn context tokens onto the SingleResult", async () => { From fcf389ae72c6267887c4e486bccbf844fff517bb Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 02:49:33 +0000 Subject: [PATCH 036/205] fix(coding-agent): deferred ask timeout until display Started the ask tool fallback timeout from the selector presentation callback so queued dialogs do not consume the user's response window. Refs #4995 --- .../src/extensibility/extensions/types.ts | 2 + .../src/modes/components/hook-selector.ts | 2 + .../controllers/extension-ui-controller.ts | 1 + packages/coding-agent/src/tools/ask.ts | 5 +- .../coding-agent/test/ask-timeout.test.ts | 74 ++++++++++++++++++- 5 files changed, 79 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 59b67e3b9..aa2f0764e 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -125,6 +125,8 @@ export interface ExtensionUIDialogOptions { timeout?: number; /** Invoked when the UI times out while waiting for a selection/input */ onTimeout?: () => void; + /** Invoked when the UI-managed timeout countdown starts */ + onTimeoutStart?: () => void; /** Invoked when user input resets a UI-managed timeout countdown */ onTimeoutReset?: () => void; /** Initial cursor position for select dialogs (0-indexed) */ diff --git a/packages/coding-agent/src/modes/components/hook-selector.ts b/packages/coding-agent/src/modes/components/hook-selector.ts index f5e5570b3..3165166ae 100644 --- a/packages/coding-agent/src/modes/components/hook-selector.ts +++ b/packages/coding-agent/src/modes/components/hook-selector.ts @@ -61,6 +61,7 @@ export interface HookSelectorOptions { tui?: TUI; timeout?: number; onTimeout?: () => void; + onTimeoutStart?: () => void; onTimeoutReset?: () => void; initialIndex?: number; outline?: boolean; @@ -235,6 +236,7 @@ export class HookSelectorComponent extends Container { } if (opts?.timeout && opts.timeout > 0 && opts.tui) { + opts.onTimeoutStart?.(); this.#countdown = new CountdownTimer( opts.timeout, opts.tui, diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index ef27b3644..f21932721 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -624,6 +624,7 @@ export class ExtensionUiController { initialIndex: dialogOptions?.initialIndex, timeout: dialogOptions?.timeout, onTimeout: dialogOptions?.onTimeout, + onTimeoutStart: dialogOptions?.onTimeoutStart, onTimeoutReset: dialogOptions?.onTimeoutReset, tui: this.ctx.ui, outline: dialogOptions?.outline, diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index 15fdc8994..105284b38 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -389,6 +389,7 @@ interface UIContext { signal?: AbortSignal; outline?: boolean; onTimeout?: () => void; + onTimeoutStart?: () => void; onTimeoutReset?: () => void; onLeft?: () => void; onRight?: () => void; @@ -455,6 +456,7 @@ async function askSingleQuestion( signal: dialogSignal, outline: true, onTimeout, + onTimeoutStart: timeoutMs === undefined ? undefined : () => armFallbackTimeout(timeoutMs), onTimeoutReset: timeoutMs === undefined ? undefined : () => armFallbackTimeout(timeoutMs), helpText, selectionMarker: marker?.selectionMarker, @@ -471,9 +473,6 @@ async function askSingleQuestion( } : undefined, }; - if (timeoutMs !== undefined) { - armFallbackTimeout(timeoutMs); - } try { const choice = dialogSignal ? await untilAborted(dialogSignal, () => ui.select(prompt, optionsToShow, dialogOptions)) diff --git a/packages/coding-agent/test/ask-timeout.test.ts b/packages/coding-agent/test/ask-timeout.test.ts index 34a153207..be8e2ba8b 100644 --- a/packages/coding-agent/test/ask-timeout.test.ts +++ b/packages/coding-agent/test/ask-timeout.test.ts @@ -48,7 +48,10 @@ describe("AskTool timeout", () => { it("auto-selects the recommended option when the selector does not settle", async () => { vi.useFakeTimers(); - const select = vi.fn(() => new Promise(() => {})); + const select = vi.fn((_title, _options, dialogOptions) => { + dialogOptions?.onTimeoutStart?.(); + return new Promise(() => {}); + }); const abort = vi.fn(); const context = { hasUI: true, @@ -101,6 +104,7 @@ describe("AskTool timeout", () => { vi.useFakeTimers(); let resetTimeout: (() => void) | undefined; const select = vi.fn((_title, _options, dialogOptions) => { + dialogOptions?.onTimeoutStart?.(); resetTimeout = dialogOptions?.onTimeoutReset; return new Promise(() => {}); }); @@ -161,17 +165,83 @@ describe("AskTool timeout", () => { expect(abort).not.toHaveBeenCalled(); }); - it("notifies callers when selector input resets the UI countdown", () => { + it("does not run the fallback timeout while the selector is queued", async () => { vi.useFakeTimers(); + let startTimeout: (() => void) | undefined; + const select = vi.fn((_title, _options, dialogOptions) => { + startTimeout = dialogOptions?.onTimeoutStart; + return new Promise(() => {}); + }); + const abort = vi.fn(); + const context = { + hasUI: true, + ui: { + select, + editor: vi.fn(), + }, + abort, + } as unknown as AgentToolContext; + let result: AskExecutionResult | undefined; + let rejection: unknown; + + void createAskTool() + .execute( + "ask-timeout", + { + questions: [ + { + id: "db", + question: "Which database?", + options: [{ label: "SQLite" }, { label: "Postgres" }], + recommended: 1, + }, + ], + }, + undefined, + undefined, + context, + ) + .then( + value => { + result = value; + }, + error => { + rejection = error; + }, + ); + + await drainMicrotasks(); + expect(startTimeout).toBeDefined(); + + vi.advanceTimersByTime(10); + await drainMicrotasks(); + + expect(result).toBeUndefined(); + + startTimeout?.(); + vi.advanceTimersByTime(10); + await drainMicrotasks(); + + expect(rejection).toBeUndefined(); + expect(result?.details?.selectedOptions).toEqual(["Postgres"]); + expect(result?.details?.timedOut).toBe(true); + expect(abort).not.toHaveBeenCalled(); + }); + + it("notifies callers when the selector countdown starts and resets", () => { + vi.useFakeTimers(); + const onTimeoutStart = vi.fn(); const onTimeoutReset = vi.fn(); const selector = new HookSelectorComponent("Pick one", ["SQLite", "Postgres"], vi.fn(), vi.fn(), { timeout: 10, tui: { requestRender: vi.fn() } as unknown as TUI, + onTimeoutStart, onTimeoutReset, }); selector.handleInput("j"); + expect(onTimeoutStart).toHaveBeenCalledTimes(1); expect(onTimeoutReset).toHaveBeenCalledTimes(1); selector.dispose(); }); From 095353ed43f64daa85a8a2b0d3bb7052ad905035 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 03:05:54 +0000 Subject: [PATCH 037/205] fix(coding-agent): preserved multi-question timeout defaults Applied the timeout auto-selection before multi-question single-choice prompts auto-advance, and replaced pending test promises with Promise.withResolvers(). Refs #4995 --- packages/coding-agent/src/tools/ask.ts | 3 + .../coding-agent/test/ask-timeout.test.ts | 75 ++++++++++++++++++- 2 files changed, 75 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index 105284b38..26b9c9d67 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -642,6 +642,9 @@ async function askSingleQuestion( customInput = undefined; break; } + if (timedOut && selectedOptions.length === 0 && customInput === undefined) { + selectedOptions = getAutoSelectionOnTimeout(questionOptions, recommended); + } if (navigation?.allowForward) { return { selectedOptions, customInput, timedOut, navigation: "forward" }; } diff --git a/packages/coding-agent/test/ask-timeout.test.ts b/packages/coding-agent/test/ask-timeout.test.ts index be8e2ba8b..bd2bf36c7 100644 --- a/packages/coding-agent/test/ask-timeout.test.ts +++ b/packages/coding-agent/test/ask-timeout.test.ts @@ -50,7 +50,7 @@ describe("AskTool timeout", () => { vi.useFakeTimers(); const select = vi.fn((_title, _options, dialogOptions) => { dialogOptions?.onTimeoutStart?.(); - return new Promise(() => {}); + return Promise.withResolvers().promise; }); const abort = vi.fn(); const context = { @@ -106,7 +106,7 @@ describe("AskTool timeout", () => { const select = vi.fn((_title, _options, dialogOptions) => { dialogOptions?.onTimeoutStart?.(); resetTimeout = dialogOptions?.onTimeoutReset; - return new Promise(() => {}); + return Promise.withResolvers().promise; }); const abort = vi.fn(); const context = { @@ -170,7 +170,7 @@ describe("AskTool timeout", () => { let startTimeout: (() => void) | undefined; const select = vi.fn((_title, _options, dialogOptions) => { startTimeout = dialogOptions?.onTimeoutStart; - return new Promise(() => {}); + return Promise.withResolvers().promise; }); const abort = vi.fn(); const context = { @@ -228,6 +228,75 @@ describe("AskTool timeout", () => { expect(abort).not.toHaveBeenCalled(); }); + it("auto-selects timed-out single-choice questions before advancing multi-question asks", async () => { + vi.useFakeTimers(); + let callCount = 0; + const select = vi.fn((_title, _options, dialogOptions) => { + callCount += 1; + if (callCount === 1) { + dialogOptions?.onTimeoutStart?.(); + return Promise.withResolvers().promise; + } + return Promise.resolve("OAuth"); + }); + const abort = vi.fn(); + const context = { + hasUI: true, + ui: { + select, + editor: vi.fn(), + }, + abort, + } as unknown as AgentToolContext; + let result: AskExecutionResult | undefined; + let rejection: unknown; + + void createAskTool() + .execute( + "ask-timeout", + { + questions: [ + { + id: "db", + question: "Which database?", + options: [{ label: "SQLite" }, { label: "Postgres" }], + recommended: 1, + }, + { + id: "auth", + question: "Which auth?", + options: [{ label: "JWT" }, { label: "OAuth" }], + recommended: 0, + }, + ], + }, + undefined, + undefined, + context, + ) + .then( + value => { + result = value; + }, + error => { + rejection = error; + }, + ); + + await drainMicrotasks(); + vi.advanceTimersByTime(10); + await drainMicrotasks(); + await drainMicrotasks(); + + expect(rejection).toBeUndefined(); + expect(select).toHaveBeenCalledTimes(2); + expect(result?.details?.results?.[0]?.selectedOptions).toEqual(["Postgres"]); + expect(result?.details?.results?.[0]?.timedOut).toBe(true); + expect(result?.details?.results?.[1]?.selectedOptions).toEqual(["OAuth"]); + expect(result?.details?.results?.[1]?.timedOut).toBeUndefined(); + expect(abort).not.toHaveBeenCalled(); + }); + it("notifies callers when the selector countdown starts and resets", () => { vi.useFakeTimers(); const onTimeoutStart = vi.fn(); From 9af5a28c64efb6c7ea3d7e39ef71e6d15995d75e Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 03:37:26 +0000 Subject: [PATCH 038/205] fix(agent): marked terminal yield before abort - Marked terminal yield state in the synchronous tool-result hook before aborting the loop. - Ignored the later tool_execution_end event for synchronously terminated yield calls so stale events cannot suppress the next prompt. Fixes #4963 --- .../coding-agent/src/session/agent-session.ts | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 9fbe9f3a0..d2bd9310d 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1846,6 +1846,7 @@ export class AgentSession { * Cleared before every new prompt turn so the next turn evaluates cleanly. */ #yieldTerminationPending = false; + #synchronouslyTerminatedYieldToolCallIds = new Set(); #providerSessionState = new Map(); #hindsightSessionState: HindsightSessionState | undefined = undefined; readonly rawSseDebugBuffer: RawSseDebugBuffer; @@ -3672,9 +3673,11 @@ export class AgentSession { } } if (event.type === "tool_execution_end" && this.#isTerminalYieldToolResult(event)) { - this.#lastSuccessfulYieldToolCallId = event.toolCallId; - this.#yieldTerminationPending = true; - this.agent.abort(); + const alreadyTerminated = this.#synchronouslyTerminatedYieldToolCallIds.delete(event.toolCallId); + if (!alreadyTerminated) { + this.#markTerminalYieldToolCall(event.toolCallId); + this.agent.abort(); + } } // TTSR: Check for pattern matches on assistant text/thinking and tool argument deltas @@ -4427,6 +4430,8 @@ export class AgentSession { result: ctx.result, }) ) { + this.#markTerminalYieldToolCall(ctx.toolCall.id); + this.#synchronouslyTerminatedYieldToolCallIds.add(ctx.toolCall.id); this.agent.abort(); } return this.#ttsrAfterToolCall(ctx); @@ -10616,6 +10621,11 @@ export class AgentSession { ); } + #markTerminalYieldToolCall(toolCallId: string): void { + this.#lastSuccessfulYieldToolCallId = toolCallId; + this.#yieldTerminationPending = true; + } + #assistantMessageHasSuccessfulYieldToolCall(assistantMessage: AssistantMessage, toolCallId: string): boolean { const lastToolCall = assistantMessage.content .slice() From 9002f4ff0aaef14e3c9ada238201a5aeb0fe9f19 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 03:45:36 +0000 Subject: [PATCH 039/205] fix(bash): avoided aborting native timeout signal Prevented explicit bash timeouts from also aborting the AbortSignal passed to pi-natives while streamed output is still draining. Native timeout_ms now owns cancellation, and the JavaScript timer only reports the fallback timeout result. Added regression coverage for streamed output before an explicit timeout. Fixes #5021 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../coding-agent/src/exec/bash-executor.ts | 9 +++++++- .../coding-agent/test/bash-executor.test.ts | 21 +++++++++++-------- 3 files changed, 24 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..2c7330a93 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Windows bash tool crashes when an explicit timeout fires while a piped command is still streaming output; the JavaScript fallback now reports the timeout without also aborting the native timeout signal. ([#5021](https://github.com/can1357/oh-my-pi/issues/5021)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/exec/bash-executor.ts b/packages/coding-agent/src/exec/bash-executor.ts index fc7d7dc69..505feca96 100644 --- a/packages/coding-agent/src/exec/bash-executor.ts +++ b/packages/coding-agent/src/exec/bash-executor.ts @@ -300,9 +300,16 @@ export async function executeBash(command: string, options?: BashExecutorOptions const requestedTimeoutMs = options?.timeout; const deadlineTimeoutMs = requestedTimeoutMs === 0 ? undefined : Math.max(1_000, requestedTimeoutMs ?? 300_000); const nativeTimeoutMs = requestedTimeoutMs !== undefined && requestedTimeoutMs > 0 ? requestedTimeoutMs : undefined; + const nativeOwnsTimeout = nativeTimeoutMs !== undefined; if (deadlineTimeoutMs !== undefined) { timeoutTimer = setTimeout(() => { - abortCurrentExecution(); + // Explicit timeouts are already enforced inside pi-natives via + // `timeoutMs`. Do not also abort the JS AbortSignal here: on Windows, + // aborting that signal while a piped command is still forwarding output + // can terminate the Bun host before the native timeout result resolves. + if (!nativeOwnsTimeout) { + abortCurrentExecution(); + } timeoutDeferred.resolve("timeout"); }, deadlineTimeoutMs); } diff --git a/packages/coding-agent/test/bash-executor.test.ts b/packages/coding-agent/test/bash-executor.test.ts index b11a15f96..f428d3ea9 100644 --- a/packages/coding-agent/test/bash-executor.test.ts +++ b/packages/coding-agent/test/bash-executor.test.ts @@ -520,11 +520,7 @@ exit 64 expect(next.output.trim()).toBe("still_persistent"); }); - it("returns at the JavaScript timeout when native timeout cleanup stalls", async () => { - if (process.platform === "win32") { - return; - } - + it("does not abort the native signal when the JavaScript timeout fallback returns streamed output", async () => { // Compress the JS-side fallback timer (floored at 1000ms in the source) so // the safety-net fires deterministically without a real 1s wait. Only long // timers are shrunk — fs/subprocess setup keeps real scheduling — and the @@ -537,8 +533,12 @@ exit 64 ...rest, )) as typeof globalThis.setTimeout); - vi.spyOn(piNatives.Shell.prototype, "run").mockImplementation((_options, onChunk) => { - onChunk?.(null, "started\n"); + let nativeSignal: AbortSignal | undefined; + vi.spyOn(piNatives.Shell.prototype, "run").mockImplementation((options, onChunk) => { + if (options.signal instanceof AbortSignal) { + nativeSignal = options.signal; + } + onChunk?.(null, "streamed-before-timeout\n"); return Promise.withResolvers().promise; }); const abortSpy = vi.spyOn(piNatives.Shell.prototype, "abort").mockResolvedValue(); @@ -546,12 +546,15 @@ exit 64 const result = await executeBash("sleep 10", { cwd: tempDir, timeout: 1000, - sessionKey: "hung-native-timeout", + sessionKey: "explicit-timeout-keeps-native-signal", }); expect(result.cancelled).toBe(true); + expect(result.output).toContain("streamed-before-timeout"); expect(result.output).toContain("Command timed out after 1 seconds"); - expect(abortSpy).toHaveBeenCalled(); + expect(nativeSignal).toBeDefined(); + expect(nativeSignal?.aborted).toBe(false); + expect(abortSpy).not.toHaveBeenCalled(); }); it("aborts before follow-up output", async () => { From f9ead2ac43523c30122dd3064a24a01418cbf463 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 04:07:03 +0000 Subject: [PATCH 040/205] fix(ai): preserved trailing cached-token usage - Kept draining OpenAI-compatible chat-completions streams when finish_reason usage lacks cache-read fields. - Added regression coverage for vLLM-style trailing usage-only chunks carrying prompt_tokens_details.cached_tokens. Fixes #5022 --- packages/ai/CHANGELOG.md | 4 ++ .../ai/src/providers/openai-completions.ts | 38 ++++++++++++++----- .../test/openai-stream-terminal-close.test.ts | 32 ++++++++++++++++ 3 files changed, 64 insertions(+), 10 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 03dd313da..4fd35077c 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI-compatible chat-completions streams preserving vLLM-style trailing cached-token usage chunks so `cacheRead` and billable `input` session stats are accurate ([#5022](https://github.com/can1357/oh-my-pi/issues/5022)). + ## [16.3.15] - 2026-07-09 ### Breaking Changes diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 45a56d9ce..99e2e9134 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -156,6 +156,18 @@ function firstPositiveNumber(...values: unknown[]): number { return 0; } +function hasCacheReadTokenField(rawUsage: object): boolean { + const usageLike = rawUsage as OpenAICompletionsUsageLike; + if (typeof usageLike.cached_tokens === "number") return true; + if (typeof usageLike.prompt_cache_hit_tokens === "number") return true; + + const rawPromptTokenDetails = usageLike.prompt_tokens_details; + if (typeof rawPromptTokenDetails !== "object" || rawPromptTokenDetails === null) return false; + + const promptTokenDetails = rawPromptTokenDetails as OpenAICompletionsPromptTokenDetails; + return typeof promptTokenDetails.cached_tokens === "number"; +} + /** * Normalize tool call ID for Mistral. * Mistral requires tool IDs to be exactly 9 alphanumeric characters (a-z, A-Z, 0-9). @@ -990,9 +1002,18 @@ const streamOpenAICompletionsOnce = ( // Terminal-chunk bookkeeping for the post-finish grace window below. // `streamFinishedAt` flips when a chunk carries `finish_reason`; - // `sawUsagePayload` flips when a usage payload was parsed. + // `sawUsagePayload` flips when a usage payload was parsed. Some + // OpenAI-compatible servers send basic usage with `finish_reason` and + // cache-read details in a trailing usage-only chunk, so only the + // no-choice terminal path may break while those details are pending. let streamFinishedAt: number | undefined; let sawUsagePayload = false; + let awaitTrailingUsageDetails = false; + const applyUsagePayload = (rawUsage: object): void => { + output.usage = parseChunkUsage(rawUsage, model, premiumRequestsTotal); + sawUsagePayload = true; + awaitTrailingUsageDetails = !hasCacheReadTokenField(rawUsage); + }; const timedOpenaiStream = iterateWithIdleTimeout(openaiStream, { idleTimeoutMs, firstItemTimeoutMs: firstEventTimeoutMs, @@ -1030,8 +1051,7 @@ const streamOpenAICompletionsOnce = ( } if (chunk.usage) { - output.usage = parseChunkUsage(chunk.usage, model, premiumRequestsTotal); - sawUsagePayload = true; + applyUsagePayload(chunk.usage); } const choice = Array.isArray(chunk.choices) ? chunk.choices[0] : undefined; @@ -1046,8 +1066,7 @@ const streamOpenAICompletionsOnce = ( if (!chunk.usage) { const choiceUsage = (choice as OpenAICompletionsChoiceUsage).usage; if (typeof choiceUsage === "object" && choiceUsage !== null) { - output.usage = parseChunkUsage(choiceUsage, model, premiumRequestsTotal); - sawUsagePayload = true; + applyUsagePayload(choiceUsage); } } @@ -1239,11 +1258,10 @@ const streamOpenAICompletionsOnce = ( } } - // `finish_reason` + usage both observed: the chat-completions - // contract has nothing left to deliver. Break instead of waiting - // for `[DONE]`/connection close so hosts that hold the socket open - // can't park the turn until the idle watchdog errors it out. - if (streamFinishedAt !== undefined && sawUsagePayload) break; + // If usage arrived on the finish chunk without cache-read fields, + // keep draining through the grace window for vLLM-style trailing + // usage details instead of finalizing the incomplete accounting. + if (streamFinishedAt !== undefined && sawUsagePayload && !awaitTrailingUsageDetails) break; } if (streamMarkupHealing) { diff --git a/packages/ai/test/openai-stream-terminal-close.test.ts b/packages/ai/test/openai-stream-terminal-close.test.ts index 8555932c6..6f867d3a0 100644 --- a/packages/ai/test/openai-stream-terminal-close.test.ts +++ b/packages/ai/test/openai-stream-terminal-close.test.ts @@ -88,6 +88,38 @@ describe("terminal frame without connection close", () => { expect(Date.now() - startedAt).toBeLessThan(2_000); }, 10_000); + it("openai-completions: keeps reading usage-only cache details after finish usage", async () => { + const fetchMock = createNeverClosingFetch([ + completionChunk({ + choices: [{ index: 0, delta: { role: "assistant", content: "Hello" }, finish_reason: "stop" }], + usage: { prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 }, + }), + completionChunk({ + choices: [], + usage: { + prompt_tokens: 10, + completion_tokens: 5, + total_tokens: 15, + prompt_tokens_details: { cached_tokens: 4 }, + }, + }), + ]); + + const startedAt = Date.now(); + const result = await streamOpenAICompletions(completionsModel, baseContext(), { + apiKey: "test-key", + fetch: fetchMock, + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(result.errorMessage).toBeUndefined(); + expect(result.content).toEqual([{ type: "text", text: "Hello" }]); + expect(result.usage.input).toBe(6); + expect(result.usage.cacheRead).toBe(4); + expect(result.usage.output).toBe(5); + expect(Date.now() - startedAt).toBeLessThan(2_000); + }, 10_000); + it("openai-completions: ends cleanly via the grace window when no usage chunk ever arrives", async () => { const fetchMock = createNeverClosingFetch([ completionChunk({ choices: [{ index: 0, delta: { role: "assistant", content: "Hello" } }] }), From bc9f4c4be64665f6c51f5353e3718969ab4be64f Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 04:13:18 +0000 Subject: [PATCH 041/205] fix(coding-agent): armed ask timeout for legacy ui Added an explicit timeout presentation capability so interactive queued dialogs defer the fallback while older UI implementations still get an immediate tool-owned timeout. Refs #4995 --- .../src/extensibility/extensions/types.ts | 2 ++ .../modes/controllers/extension-ui-controller.ts | 1 + packages/coding-agent/src/tools/ask.ts | 13 ++++++++++--- packages/coding-agent/test/ask-timeout.test.ts | 6 ++---- 4 files changed, 15 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index aa2f0764e..e3bb89739 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -180,6 +180,8 @@ export type AutocompleteProviderFactory = (current: AutocompleteProvider) => Aut // and may be invoked from event handlers that have already taken the agent // loop's lock — hooks intentionally cannot. export interface ExtensionUIContext { + /** True when selector timeouts start only after the dialog is presented. */ + timeoutStartsOnPresentation?: boolean; /** Show a selector and return the selected label, even when an option also includes a description. */ select( title: string, diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index f21932721..99a840721 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -61,6 +61,7 @@ export class ExtensionUiController { async initHooksAndCustomTools(): Promise { // Create and set hook & tool UI context const uiContext: ExtensionUIContext = { + timeoutStartsOnPresentation: true, select: (title, options, dialogOptions) => this.showCollabAwareSelector(title, options, dialogOptions), confirm: (title, message, _dialogOptions) => this.showHookConfirm(title, message), input: (title, placeholder, dialogOptions) => this.showHookInput(title, placeholder, dialogOptions), diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index 26b9c9d67..efaf65db5 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -380,6 +380,7 @@ interface AskSingleQuestionOptions { } interface UIContext { + timeoutStartsOnPresentation?: boolean; select( prompt: string, options: ExtensionUISelectItem[], @@ -474,9 +475,14 @@ async function askSingleQuestion( : undefined, }; try { - const choice = dialogSignal - ? await untilAborted(dialogSignal, () => ui.select(prompt, optionsToShow, dialogOptions)) - : await ui.select(prompt, optionsToShow, dialogOptions); + const runSelect = () => { + const selection = ui.select(prompt, optionsToShow, dialogOptions); + if (timeoutMs !== undefined && !ui.timeoutStartsOnPresentation) { + armFallbackTimeout(timeoutMs); + } + return selection; + }; + const choice = dialogSignal ? await untilAborted(dialogSignal, runSelect) : await runSelect(); if (!timeoutTriggered && choice === undefined && typeof timeout === "number") { // Fallback for UI surfaces that enforce `timeout` without invoking // `onTimeout`: their auto-cancel resolves right at the deadline. A @@ -774,6 +780,7 @@ export class AskTool implements AgentTool { const extensionUi = context.ui; const ui: UIContext = { + timeoutStartsOnPresentation: extensionUi.timeoutStartsOnPresentation, select: (prompt, options, dialogOptions) => extensionUi.select(prompt, options, dialogOptions), editor: (title, prefill, dialogOptions, editorOptions) => extensionUi.editor(title, prefill, dialogOptions, editorOptions), diff --git a/packages/coding-agent/test/ask-timeout.test.ts b/packages/coding-agent/test/ask-timeout.test.ts index bd2bf36c7..0caaf8bd6 100644 --- a/packages/coding-agent/test/ask-timeout.test.ts +++ b/packages/coding-agent/test/ask-timeout.test.ts @@ -48,10 +48,7 @@ describe("AskTool timeout", () => { it("auto-selects the recommended option when the selector does not settle", async () => { vi.useFakeTimers(); - const select = vi.fn((_title, _options, dialogOptions) => { - dialogOptions?.onTimeoutStart?.(); - return Promise.withResolvers().promise; - }); + const select = vi.fn(() => Promise.withResolvers().promise); const abort = vi.fn(); const context = { hasUI: true, @@ -176,6 +173,7 @@ describe("AskTool timeout", () => { const context = { hasUI: true, ui: { + timeoutStartsOnPresentation: true, select, editor: vi.fn(), }, From 8495931e0373aa723bcc62aba71881fc61647360 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 04:23:54 +0000 Subject: [PATCH 042/205] fix(ai): preserved zero cache placeholder usage - Required positive cache-read fields before treating terminal usage accounting as complete. - Covered zero placeholder usage followed by trailing cached-token details. Fixes #5022 --- packages/ai/src/providers/openai-completions.ts | 10 +++++----- packages/ai/test/openai-stream-terminal-close.test.ts | 9 +++++++-- 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 99e2e9134..f200c268c 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -156,16 +156,16 @@ function firstPositiveNumber(...values: unknown[]): number { return 0; } -function hasCacheReadTokenField(rawUsage: object): boolean { +function hasPositiveCacheReadTokenField(rawUsage: object): boolean { const usageLike = rawUsage as OpenAICompletionsUsageLike; - if (typeof usageLike.cached_tokens === "number") return true; - if (typeof usageLike.prompt_cache_hit_tokens === "number") return true; + if (typeof usageLike.cached_tokens === "number" && usageLike.cached_tokens > 0) return true; + if (typeof usageLike.prompt_cache_hit_tokens === "number" && usageLike.prompt_cache_hit_tokens > 0) return true; const rawPromptTokenDetails = usageLike.prompt_tokens_details; if (typeof rawPromptTokenDetails !== "object" || rawPromptTokenDetails === null) return false; const promptTokenDetails = rawPromptTokenDetails as OpenAICompletionsPromptTokenDetails; - return typeof promptTokenDetails.cached_tokens === "number"; + return typeof promptTokenDetails.cached_tokens === "number" && promptTokenDetails.cached_tokens > 0; } /** @@ -1012,7 +1012,7 @@ const streamOpenAICompletionsOnce = ( const applyUsagePayload = (rawUsage: object): void => { output.usage = parseChunkUsage(rawUsage, model, premiumRequestsTotal); sawUsagePayload = true; - awaitTrailingUsageDetails = !hasCacheReadTokenField(rawUsage); + awaitTrailingUsageDetails = !hasPositiveCacheReadTokenField(rawUsage); }; const timedOpenaiStream = iterateWithIdleTimeout(openaiStream, { idleTimeoutMs, diff --git a/packages/ai/test/openai-stream-terminal-close.test.ts b/packages/ai/test/openai-stream-terminal-close.test.ts index 6f867d3a0..254ef4d2c 100644 --- a/packages/ai/test/openai-stream-terminal-close.test.ts +++ b/packages/ai/test/openai-stream-terminal-close.test.ts @@ -88,11 +88,16 @@ describe("terminal frame without connection close", () => { expect(Date.now() - startedAt).toBeLessThan(2_000); }, 10_000); - it("openai-completions: keeps reading usage-only cache details after finish usage", async () => { + it("openai-completions: ignores zero cache placeholder until trailing positive cache details arrive", async () => { const fetchMock = createNeverClosingFetch([ completionChunk({ choices: [{ index: 0, delta: { role: "assistant", content: "Hello" }, finish_reason: "stop" }], - usage: { prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 }, + usage: { + prompt_tokens: 10, + completion_tokens: 5, + total_tokens: 15, + prompt_tokens_details: { cached_tokens: 0 }, + }, }), completionChunk({ choices: [], From cdecf65f8b8d8ce027bfbef639a359ac209afb1d Mon Sep 17 00:00:00 2001 From: usr_bin_roygbiv Date: Fri, 10 Jul 2026 00:40:52 -0500 Subject: [PATCH 043/205] fix(compaction): retry after AWS credential failures --- packages/ai/src/error/flags.ts | 5 +- packages/ai/test/error-aierr.test.ts | 5 ++ packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/session/agent-session.ts | 16 +---- .../compaction-prefer-current-model.test.ts | 71 +++++++++++++++++++ 5 files changed, 86 insertions(+), 15 deletions(-) diff --git a/packages/ai/src/error/flags.ts b/packages/ai/src/error/flags.ts index a61638ea1..f6570b285 100644 --- a/packages/ai/src/error/flags.ts +++ b/packages/ai/src/error/flags.ts @@ -1,5 +1,6 @@ import { isUnexpectedSocketCloseMessage } from "@oh-my-pi/pi-utils"; import type { Api, AssistantMessage } from "../types"; +import { AwsCredentialsError } from "./aws"; import { AnthropicConnectionError, AnthropicConnectionTimeoutError, @@ -346,7 +347,9 @@ export function classify(error: unknown, api?: Api): number { } } - if (link instanceof AnthropicConnectionTimeoutError) { + if (link instanceof AwsCredentialsError) { + kinds |= Flag.AuthFailed; + } else if (link instanceof AnthropicConnectionTimeoutError) { kinds |= Flag.Timeout | Flag.Transient; } else if (link instanceof AnthropicConnectionError) { kinds |= Flag.Transient; diff --git a/packages/ai/test/error-aierr.test.ts b/packages/ai/test/error-aierr.test.ts index 810a76f87..57617b60e 100644 --- a/packages/ai/test/error-aierr.test.ts +++ b/packages/ai/test/error-aierr.test.ts @@ -32,6 +32,11 @@ describe("AIError.classify — structural provider errors", () => { ).toBe(true); }); + it("classifies a typed AWS credential-resolution failure as authFailed", () => { + const id = AIError.classify(new AIError.AwsCredentialsError("opaque provider setup failure", "resolution")); + expect(AIError.is(id, AIError.Flag.AuthFailed)).toBe(true); + }); + it("maps the usage_limit_reached code to usageLimit on a 429", () => { const id = AIError.classify( new AIError.ProviderHttpError("Payment Required", 429, { code: "usage_limit_reached" }), diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..30578c75c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed compaction aborting instead of trying an authenticated fallback model when Amazon Bedrock credential resolution fails before a request is sent. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..2654751e3 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -11921,18 +11921,6 @@ export class AgentSession { return candidates; } - #isCompactionAuthFailure(error: unknown): boolean { - if (!(error instanceof Error)) return false; - // Real provider 401/403 — surfaced as `.status` by the compaction layer - // (see `createSummarizationError` in packages/agent/src/compaction/compaction.ts). - // Without this branch, an expired/revoked Anthropic key would bypass the - // authenticated-fallback path and dump the raw HTTP body into the UI. - const status = (error as Error & { status?: number }).status; - if (status === 401 || status === 403) return true; - // pi-native gateway synthetic for "no credential configured" (issue #986). - // Carries no HTTP status, so the legacy message regex stays. - return /auth_unavailable|no auth available/i.test(error.message); - } #buildCompactionAuthError(): Error { const currentModel = this.model; @@ -11997,7 +11985,7 @@ export class AgentSession { }, ); } catch (error) { - if (!this.#isCompactionAuthFailure(error)) { + if (!AIError.is(AIError.classify(error, candidate.api), AIError.Flag.AuthFailed)) { throw error; } } @@ -12689,7 +12677,7 @@ export class AgentSession { const message = error instanceof Error ? error.message : String(error); const id = AIError.classify(error, candidate.api); - if (this.#isCompactionAuthFailure(error)) { + if (AIError.is(id, AIError.Flag.AuthFailed)) { lastError = this.#buildCompactionAuthError(); break; } diff --git a/packages/coding-agent/test/compaction-prefer-current-model.test.ts b/packages/coding-agent/test/compaction-prefer-current-model.test.ts index 0bceded5f..9b6b5adc9 100644 --- a/packages/coding-agent/test/compaction-prefer-current-model.test.ts +++ b/packages/coding-agent/test/compaction-prefer-current-model.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction"; +import * as AIError from "@oh-my-pi/pi-ai/error"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -99,6 +100,76 @@ describe("compaction prefers the current session model over modelRoles.default", expect(`${firstCandidate.provider}/${firstCandidate.id}`).toBe(`${currentModel.provider}/${currentModel.id}`); }); + it("falls back when the authenticated Bedrock candidate cannot resolve AWS credentials", async () => { + const currentModel = getBundledModel("amazon-bedrock", "global.anthropic.claude-opus-4-6-v1"); + const fallbackModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!currentModel || !fallbackModel) { + throw new Error("Expected bundled test models to exist"); + } + + const settings = Settings.isolated({ "compaction.keepRecentTokens": 1, "compaction.strategy": "context-full" }); + settings.setModelRole("smol", `${fallbackModel.provider}/${fallbackModel.id}`); + + const agent = new Agent({ + initialState: { + model: currentModel, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }); + + authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey(currentModel.provider, "bedrock-credentials"); + authStorage.setRuntimeApiKey(fallbackModel.provider, "anthropic-token"); + modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + session.subscribe(() => {}); + + for (const [userText, assistantText] of [ + ["first question", "first answer"], + ["second question", "second answer"], + ] as const) { + const user = userMsg(userText); + const assistant = assistantMsg(assistantText); + session.agent.appendMessage(user); + session.sessionManager.appendMessage(user); + session.agent.appendMessage(assistant); + session.sessionManager.appendMessage(assistant); + } + + const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => { + if (model.provider === currentModel.provider && model.id === currentModel.id) { + throw new AIError.AwsCredentialsError("opaque provider setup failure", "resolution"); + } + if (model.provider !== fallbackModel.provider || model.id !== fallbackModel.id) { + throw new Error(`Unexpected compaction model ${model.provider}/${model.id}`); + } + return { + summary: "fallback summary", + shortSummary: "fallback short summary", + firstKeptEntryId: preparation.firstKeptEntryId, + tokensBefore: 42, + details: { provider: model.provider }, + }; + }); + + const result = await session.compact(); + + expect(result.summary).toBe("fallback summary"); + expect(compactSpy).toHaveBeenCalledTimes(2); + expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([ + `${currentModel.provider}/${currentModel.id}`, + `${fallbackModel.provider}/${fallbackModel.id}`, + ]); + }); + it("uses compactionModel only for the summary call and leaves the active model unchanged", async () => { const baseCurrentModel = getBundledModel("anthropic", "claude-sonnet-4-5"); const compactionModel = getBundledModel("openai", "gpt-5"); From 6db9ab606a6db7d141ac2189d9b192f4f4f39856 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 06:00:31 +0000 Subject: [PATCH 044/205] fix(tui): prevented destructive paint flicker Replayed destructive full paints from home without issuing ED2 before the transcript rows, so resume and resize settles do not expose a blank viewport on terminals without synchronized output. Kept ED3 as the single native-history clear path and updated renderer regression coverage for the new byte contract. Fixes #5028 --- docs/tui-core-renderer.md | 10 ++++--- packages/tui/CHANGELOG.md | 4 +++ packages/tui/src/tui.ts | 27 +++++++++++-------- packages/tui/test/issue-2115-repro.test.ts | 3 ++- packages/tui/test/render-regressions.test.ts | 7 ++--- .../tui/test/resize-viewport-defer.test.ts | 5 +++- 6 files changed, 36 insertions(+), 20 deletions(-) diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md index 10fe737ce..25dfb4ad0 100644 --- a/docs/tui-core-renderer.md +++ b/docs/tui-core-renderer.md @@ -110,7 +110,7 @@ updates never rewrite anything a scrolled reader could be looking at. | Emitter | Bytes | When | |---|---|---| -| `#emitFullPaint` | clears + `frame[0, C')` + window rows | gestures only. `clearScrollback` ⇒ `\x1b[2J\x1b[H\x1b[3J`; otherwise ED22 (when supported) + `\x1b[2J\x1b[H` | +| `#emitFullPaint` | home + `frame[0, C')` + window rows; with `clearScrollback`, ED3 clears history without an ED2 viewport blank | gestures only | | `#emitUpdate` scroll-append | `\r\n` + new bottom rows + changed-row range | the rows leaving the screen are exactly the chunk, content untouched since painted | | `#emitUpdate` in-window diff | relative move + changed-row range rewrite | nothing scrolls, nothing commits (cursor-only when nothing changed) | | `#emitUpdate` seam rewrite | chunk rows + full window rewrite | commit advance, window re-anchor, hidden-gap backfill, mux resize | @@ -118,9 +118,11 @@ updates never rewrite anything a scrolled reader could be looking at. **ED3 (`CSI 3 J`) is emitted in exactly one place** — `#emitFullPaint` with `clearScrollback: true` — and is reached only by user gestures: session replace/branch/resume (`requestRender(true, { clearScrollback: true })`), -resize outside a multiplexer, `resetDisplay()` (Ctrl+L). A gesture pins the -user to the tail, so the snap is acceptable; multiplexers never get ED3 (it is -a no-op there and a replay would duplicate pane history). +resize outside a multiplexer, `resetDisplay()` (Ctrl+L). It clears native +history without `ED2` first; the replay overwrites every row from home so +terminals without synchronized output do not expose a blank viewport. A gesture +pins the user to the tail, so the history snap is acceptable; multiplexers never +get ED3 (it is a no-op there and a replay would duplicate pane history). The ordinary update path never emits ED2/ED3 or an absolute cursor home — several terminal families snap a scrolled reader to the bottom on those. diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index cb2b07009..8106893df 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed resume/session-replace and resize-settle full paints blanking the live viewport before replaying the transcript, preventing flicker on terminals without effective synchronized output ([#5028](https://github.com/can1357/oh-my-pi/issues/5028)). + ## [16.3.14] - 2026-07-09 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index a916af1d5..955b0e39a 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -561,10 +561,9 @@ export class Container implements Component { * method owns the bytes written and the state update. * * - `fullPaint`: gesture-driven replay — initial paint, session replacement, - * resize, resetDisplay. Clears the viewport and (for destructive replaces, - * outside multiplexers) native scrollback via ED3, then writes the - * committed prefix and the visible window. The only ED3 callsite in the - * engine. + * resize, resetDisplay. Rewrites the frame from home; destructive replaces + * clear native scrollback via ED3 without first blanking the viewport. The + * only ED3 callsite in the engine. * - `update`: ordinary frame. Commits the newly settled chunk at the * scrollback seam (if any) and repaints the window with relative moves. */ @@ -3221,7 +3220,10 @@ export class TUI extends Container { } let buffer = this.#paintBeginSequence + this.#leaveResizeAltSequence() + purgeSequence; if (options.clearScrollback) { - buffer += "\x1b[2J\x1b[H\x1b[3J"; + // Clear native history without blanking the live viewport first. The + // replay below rewrites every visible row from home, including blanks, + // so terminals without DEC 2026 never expose an ED2-cleared frame. + buffer += "\x1b[H\x1b[3J"; } else { // Best-effort: push the pre-paint screen into scrollback on // terminals that implement kitty's ED 22 @@ -3254,21 +3256,24 @@ export class TUI extends Container { if (paintLines === null) { // Common path: emit straight from the source arrays (the // pre-merge two-loop form); byte-identical to replaying the - // merged array. + // merged array. Destructive history clears deliberately avoid ED2, so + // each row must self-clear stale cells left by the previous viewport. for (let i = 0; i < chunkTo; i++) { if (i > 0) buffer += "\r\n"; - buffer += this.#terminalLine(frame[i] ?? ""); + buffer += options.clearScrollback + ? this.#lineRewriteSequence(frame[i] ?? "", width) + : this.#terminalLine(frame[i] ?? ""); } for (let screenRow = 0; screenRow < height; screenRow++) { if (chunkTo + screenRow > 0) buffer += "\r\n"; - buffer += this.#terminalLine(visibleTexts ? (visibleTexts[screenRow] ?? "") : (window[screenRow] ?? "")); + const line = visibleTexts ? (visibleTexts[screenRow] ?? "") : (window[screenRow] ?? ""); + buffer += options.clearScrollback ? this.#lineRewriteSequence(line, width) : this.#terminalLine(line); } } else { for (let i = 0; i < paintLines.length; i++) { if (i > 0) buffer += "\r\n"; - buffer += this.#terminalLine( - visibleTexts && i >= visibleStart ? visibleTexts[i - visibleStart] : (paintLines[i] ?? ""), - ); + const line = visibleTexts && i >= visibleStart ? visibleTexts[i - visibleStart] : (paintLines[i] ?? ""); + buffer += options.clearScrollback ? this.#lineRewriteSequence(line, width) : this.#terminalLine(line); } } buffer += fillSequence; diff --git a/packages/tui/test/issue-2115-repro.test.ts b/packages/tui/test/issue-2115-repro.test.ts index 6ff224d48..b6672f4a9 100644 --- a/packages/tui/test/issue-2115-repro.test.ts +++ b/packages/tui/test/issue-2115-repro.test.ts @@ -107,8 +107,9 @@ describe("issue #2115: ConPTY large-session resume truncates at logical lines", tui.start({ clearScrollback: true }); await term.waitForRender(); - const fullPaint = writes.find(write => write.includes("\x1b[2J")); + const fullPaint = writes.find(write => write.includes("\x1b[3J")); expect(fullPaint).toBeDefined(); + expect(fullPaint).not.toContain("\x1b[2J"); expect(Buffer.byteLength(fullPaint ?? "", "utf8")).toBeLessThan(128 * 1024); expect(fullPaint).toContain("older lines hidden"); expect(fullPaint).not.toContain("第00000行"); diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 83f8a0aa4..6d39fa08c 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -432,7 +432,7 @@ describe("TUI terminal-state regressions", () => { tui.resetDisplay(); await settle(term); - expect(writes.some(write => write.includes("\x1b[2J\x1b[H\x1b[3J"))).toBe(true); + expect(writes.some(write => write.includes("\x1b[H\x1b[3J") && !write.includes("\x1b[2J"))).toBe(true); expect(term.getScrollBuffer().map(line => line.trimEnd())).toEqual(rows("L", 8)); expect(visible(term)).toEqual(["L5", "L6", "L7"]); } finally { @@ -1357,7 +1357,7 @@ describe("TUI terminal-state regressions", () => { } }); - it("uses ED3 for destructive rebuilds even when CSI 22 J is supported", async () => { + it("uses ED3 without blanking the viewport for destructive rebuilds even when CSI 22 J is supported", async () => { const saved = TERMINAL.supportsScreenToScrollback; setTerminalScreenToScrollback(true); const term = new VirtualTerminal(20, 3); @@ -1373,7 +1373,8 @@ describe("TUI terminal-state regressions", () => { tui.requestRender(true, { clearScrollback: true }); await settle(term); const out = writes.join(""); - expect(out).toContain("\x1b[2J\x1b[H\x1b[3J"); + expect(out).toContain("\x1b[H\x1b[3J"); + expect(out).not.toContain("\x1b[2J"); expect(out).not.toContain("\x1b[22J"); } finally { tui.stop(); diff --git a/packages/tui/test/resize-viewport-defer.test.ts b/packages/tui/test/resize-viewport-defer.test.ts index a0ceac04d..622f8d212 100644 --- a/packages/tui/test/resize-viewport-defer.test.ts +++ b/packages/tui/test/resize-viewport-defer.test.ts @@ -263,7 +263,9 @@ describe("non-multiplexer resize viewport fast path", () => { await scheduler.flushImmediates(term); // Settle window elapses: exactly one authoritative full paint that - // erases native scrollback (ED3) and replays every block. + // clears native scrollback (ED3) and replays every block. It must not + // blank the live viewport with ED2 first; terminals without DEC 2026 + // expose that blank frame as resize/session-replace flicker. for (const b of blocks) b.renderCount = 0; await scheduler.flushAll(term); @@ -273,6 +275,7 @@ describe("non-multiplexer resize viewport fast path", () => { // full replay or a stray scrollback erase into the settle. expect(tui.fullRedraws).toBe(baselineFull + 1); expect(eraseScrollbackCount(writes)).toBe(1); + expect(writes.join("")).not.toContain("\x1b[2J"); // The full replay lays out the whole transcript, off-screen blocks // included. expect(blocks.every(b => b.renderCount > 0)).toBe(true); From 993b21204ec55f279b1ae15500173fa0e0eaf341 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 07:14:49 +0000 Subject: [PATCH 045/205] fix(lsp): handled go.work workspace diagnostics Detected go.work before go.mod for workspace diagnostics and expanded go build package patterns from go.work use entries. Added regression coverage for go.work-only roots and mixed go.work/go.mod workspaces. Fixes #5038 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/lsp/index.ts | 143 ++++++++++++++---- .../test/tools/lsp-regressions.test.ts | 135 ++++++++++++++++- 3 files changed, 252 insertions(+), 30 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..a3ba17858 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed LSP workspace diagnostics for Go workspaces so roots with `go.work` are recognized and every `go.work use` module is included in the `go build` package patterns. ([#5038](https://github.com/can1357/oh-my-pi/issues/5038)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/lsp/index.ts b/packages/coding-agent/src/lsp/index.ts index 360f2513c..ca829efd6 100644 --- a/packages/coding-agent/src/lsp/index.ts +++ b/packages/coding-agent/src/lsp/index.ts @@ -573,8 +573,79 @@ interface ProjectType { description: string; } +/** Convert a `go.work` use directory into the package pattern `go build` needs. */ +function goWorkspaceBuildPattern(diskPath: string): string | null { + const trimmed = diskPath.trim(); + if (!trimmed) return null; + + const isAbsolute = path.isAbsolute(trimmed) || path.win32.isAbsolute(trimmed); + const normalized = trimmed.replaceAll("\\", "/").replace(/\/+$/, ""); + const dir = normalized || "."; + if (dir === ".") return "./..."; + if (dir.endsWith("/...")) return dir; + if (isAbsolute || dir.startsWith("./") || dir.startsWith("../")) return `${dir}/...`; + return `./${dir}/...`; +} + +/** Parse `go work edit -json` output into per-module package patterns. */ +function parseGoWorkspaceBuildPatterns(output: string): string[] { + let parsed: unknown; + try { + parsed = JSON.parse(output); + } catch { + return []; + } + + if (!parsed || typeof parsed !== "object" || !("Use" in parsed) || !Array.isArray(parsed.Use)) return []; + + const patterns = new Set(); + for (const entry of parsed.Use) { + if (!entry || typeof entry !== "object" || !("DiskPath" in entry) || typeof entry.DiskPath !== "string") { + continue; + } + const pattern = goWorkspaceBuildPattern(entry.DiskPath); + if (pattern) patterns.add(pattern); + } + return [...patterns]; +} + +/** Resolve the `go build` command for a `go.work` workspace. */ +async function resolveGoWorkspaceDiagnosticsCommand(cwd: string, signal?: AbortSignal): Promise { + const fallback = ["go", "build", "./..."]; + try { + const proc = Bun.spawn(["go", "work", "edit", "-json"], { + cwd, + stdout: "pipe", + stderr: "pipe", + windowsHide: true, + }); + const abortHandler = () => { + proc.kill(); + }; + if (signal) { + signal.addEventListener("abort", abortHandler, { once: true }); + } + + try { + const [stdout] = await Promise.all([new Response(proc.stdout).text(), new Response(proc.stderr).text()]); + const exitCode = await proc.exited; + throwIfAborted(signal); + if (exitCode !== 0) return fallback; + const patterns = parseGoWorkspaceBuildPatterns(stdout); + return patterns.length > 0 ? ["go", "build", ...patterns] : fallback; + } finally { + signal?.removeEventListener("abort", abortHandler); + } + } catch { + if (signal?.aborted) { + throw new ToolAbortError(); + } + return fallback; + } +} + /** Detect project type from root markers */ -function detectProjectType(cwd: string): ProjectType { +async function detectProjectType(cwd: string, signal?: AbortSignal): Promise { // Check for Rust (Cargo.toml) if (fs.existsSync(path.join(cwd, "Cargo.toml"))) { return { type: "rust", command: ["cargo", "check", "--message-format=short"], description: "Rust (cargo check)" }; @@ -585,6 +656,15 @@ function detectProjectType(cwd: string): ProjectType { return { type: "typescript", command: ["npx", "tsc", "--noEmit"], description: "TypeScript (tsc --noEmit)" }; } + // Check for Go workspaces before single-module Go projects. + if (fs.existsSync(path.join(cwd, "go.work"))) { + return { + type: "go", + command: await resolveGoWorkspaceDiagnosticsCommand(cwd, signal), + description: "Go workspace (go build)", + }; + } + // Check for Go (go.mod) if (fs.existsSync(path.join(cwd, "go.mod"))) { return { type: "go", command: ["go", "build", "./..."], description: "Go (go build)" }; @@ -604,47 +684,52 @@ async function runWorkspaceDiagnostics( signal?: AbortSignal, ): Promise<{ output: string; projectType: ProjectType }> { throwIfAborted(signal); - const projectType = detectProjectType(cwd); + const projectType = await detectProjectType(cwd, signal); if (!projectType.command) { return { - output: `Cannot detect project type. Supported: Rust (Cargo.toml), TypeScript (tsconfig.json), Go (go.mod), Python (pyproject.toml)`, + output: `Cannot detect project type. Supported: Rust (Cargo.toml), TypeScript (tsconfig.json), Go (go.work/go.mod), Python (pyproject.toml)`, projectType, }; } - const proc = Bun.spawn(projectType.command, { - cwd, - stdout: "pipe", - stderr: "pipe", - windowsHide: true, - }); - const abortHandler = () => { - proc.kill(); - }; - if (signal) { - signal.addEventListener("abort", abortHandler, { once: true }); - } - try { - const [stdout, stderr] = await Promise.all([new Response(proc.stdout).text(), new Response(proc.stderr).text()]); - await proc.exited; - throwIfAborted(signal); - const combined = (stdout + stderr).trim(); - if (!combined) { - return { output: "No issues found", projectType }; + const proc = Bun.spawn(projectType.command, { + cwd, + stdout: "pipe", + stderr: "pipe", + windowsHide: true, + }); + const abortHandler = () => { + proc.kill(); + }; + if (signal) { + signal.addEventListener("abort", abortHandler, { once: true }); } - // Limit output length - const lines = combined.split("\n"); - if (lines.length > 50) { - return { output: `${lines.slice(0, 50).join("\n")}\n[…${lines.length - 50}ln elided…]`, projectType }; + + try { + const [stdout, stderr] = await Promise.all([ + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + ]); + await proc.exited; + throwIfAborted(signal); + const combined = (stdout + stderr).trim(); + if (!combined) { + return { output: "No issues found", projectType }; + } + // Limit output length + const lines = combined.split("\n"); + if (lines.length > 50) { + return { output: `${lines.slice(0, 50).join("\n")}\n[…${lines.length - 50}ln elided…]`, projectType }; + } + return { output: combined, projectType }; + } finally { + signal?.removeEventListener("abort", abortHandler); } - return { output: combined, projectType }; } catch (e) { if (signal?.aborted) { throw new ToolAbortError(); } return { output: `Failed to run ${projectType.command.join(" ")}: ${e}`, projectType }; - } finally { - signal?.removeEventListener("abort", abortHandler); } } diff --git a/packages/coding-agent/test/tools/lsp-regressions.test.ts b/packages/coding-agent/test/tools/lsp-regressions.test.ts index c0867da3c..872fb0540 100644 --- a/packages/coding-agent/test/tools/lsp-regressions.test.ts +++ b/packages/coding-agent/test/tools/lsp-regressions.test.ts @@ -2,7 +2,7 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import type { RenderResultOptions } from "@oh-my-pi/pi-agent-core"; +import type { AgentToolResult, RenderResultOptions } from "@oh-my-pi/pi-agent-core"; import { preloadPluginRoots } from "@oh-my-pi/pi-coding-agent/discovery/helpers"; import { LspTool } from "@oh-my-pi/pi-coding-agent/lsp"; import * as lspClient from "@oh-my-pi/pi-coding-agent/lsp/client"; @@ -20,6 +20,7 @@ import type { DeleteFile, Diagnostic, LspClient, + LspToolDetails, RenameFile, ServerConfig, SymbolInformation, @@ -43,6 +44,7 @@ import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { clampTimeout } from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts"; import * as piUtils from "@oh-my-pi/pi-utils"; import { sanitizeText, TempDir } from "@oh-my-pi/pi-utils"; +import type { Subprocess } from "bun"; import DEFAULTS from "../../src/lsp/defaults.json" with { type: "json" }; import { getLanguageFromPath } from "../../src/utils/lang-from-path"; @@ -189,6 +191,57 @@ function installFakeLsp(handler: FakeLspHandler): FakeLspServer { return server; } +type BunSpawnOptions = Bun.SpawnOptions.SpawnOptions< + Bun.SpawnOptions.Writable, + Bun.SpawnOptions.Readable, + Bun.SpawnOptions.Readable +>; + +interface BunSpawnCall { + cmd: string[]; + options?: BunSpawnOptions; +} + +interface BunSpawnOutput { + stdout?: string; + stderr?: string; + exitCode?: number; +} + +function textStream(text: string): ReadableStream { + const body = new Response(text).body; + if (!body) { + throw new Error("Failed to create text stream"); + } + return body; +} + +function completedProcess(stdout = "", stderr = "", exitCode = 0): Subprocess { + return { + pid: 12_345, + stdout: textStream(stdout), + stderr: textStream(stderr), + exited: Promise.resolve(exitCode), + kill: () => {}, + } as Subprocess; +} + +function recordBunSpawn(calls: BunSpawnCall[], outputForCommand: (cmd: string[]) => BunSpawnOutput = () => ({})): void { + vi.spyOn(Bun, "spawn").mockImplementation(((cmd: string[], options?: BunSpawnOptions) => { + const recordedCmd = [...cmd]; + calls.push({ cmd: recordedCmd, options }); + const output = outputForCommand(recordedCmd); + return completedProcess(output.stdout, output.stderr, output.exitCode); + }) as typeof Bun.spawn); +} + +function textResult(result: AgentToolResult): string { + return result.content + .filter(block => block.type === "text") + .map(block => block.text) + .join("\n"); +} + describe("lsp regressions", () => { afterEach(() => { vi.restoreAllMocks(); @@ -1057,6 +1110,86 @@ describe("lsp regressions", () => { } }); + it("treats a go.work-only root as a Go workspace for workspace diagnostics", async () => { + const tempDir = TempDir.createSync("@omp-lsp-go-work-only-"); + const spawnCalls: BunSpawnCall[] = []; + recordBunSpawn(spawnCalls, cmd => { + if (cmd.join("\0") === "go\0work\0edit\0-json") { + return { stdout: JSON.stringify({ Use: [{ DiskPath: "./service" }] }) }; + } + return {}; + }); + + try { + const serviceDir = path.join(tempDir.path(), "service"); + await fs.promises.mkdir(serviceDir, { recursive: true }); + await Bun.write(path.join(tempDir.path(), "go.work"), ["go 1.22", "", "use ./service", ""].join("\n")); + await Bun.write(path.join(serviceDir, "go.mod"), "module example.com/service\n\ngo 1.22\n"); + + const tool = new LspTool({ cwd: tempDir.path() } as ToolSession); + const result = await tool.execute("go-work-only-diagnostics", { + action: "diagnostics", + file: "*", + }); + + const buildCalls = spawnCalls.filter(call => call.cmd[0] === "go" && call.cmd[1] === "build"); + expect(buildCalls).toHaveLength(1); + expect(buildCalls[0]?.cmd).toEqual(["go", "build", "./service/..."]); + expect(buildCalls[0]?.options?.cwd).toBe(tempDir.path()); + const output = textResult(result); + expect(output).toContain("Workspace diagnostics ("); + expect(output).toContain("go build"); + expect(output).toContain("No issues found"); + expect(output).not.toContain("Cannot detect project type"); + } finally { + tempDir.removeSync(); + } + }); + + it("builds every go.work use module when go.work and go.mod coexist", async () => { + const tempDir = TempDir.createSync("@omp-lsp-go-work-before-mod-"); + const spawnCalls: BunSpawnCall[] = []; + recordBunSpawn(spawnCalls, cmd => { + if (cmd.join("\0") === "go\0work\0edit\0-json") { + return { + stdout: JSON.stringify({ + Use: [{ DiskPath: "." }, { DiskPath: "./service" }, { DiskPath: "./tools/helper" }], + }), + }; + } + return {}; + }); + + try { + const serviceDir = path.join(tempDir.path(), "service"); + const helperDir = path.join(tempDir.path(), "tools", "helper"); + await fs.promises.mkdir(serviceDir, { recursive: true }); + await fs.promises.mkdir(helperDir, { recursive: true }); + await Bun.write(path.join(tempDir.path(), "go.mod"), "module example.com/root\n\ngo 1.22\n"); + await Bun.write(path.join(serviceDir, "go.mod"), "module example.com/service\n\ngo 1.22\n"); + await Bun.write(path.join(helperDir, "go.mod"), "module example.com/helper\n\ngo 1.22\n"); + await Bun.write( + path.join(tempDir.path(), "go.work"), + ["go 1.22", "", "use (", "\t.", "\t./service", "\t./tools/helper", ")", ""].join("\n"), + ); + + const tool = new LspTool({ cwd: tempDir.path() } as ToolSession); + const result = await tool.execute("go-work-module-patterns", { + action: "diagnostics", + file: "*", + }); + + const buildCalls = spawnCalls.filter(call => call.cmd[0] === "go" && call.cmd[1] === "build"); + expect(buildCalls).toHaveLength(1); + expect(buildCalls[0]?.cmd.slice(0, 2)).toEqual(["go", "build"]); + expect(buildCalls[0]?.cmd.slice(2).sort()).toEqual(["./...", "./service/...", "./tools/helper/..."]); + expect(textResult(result)).toContain("Workspace diagnostics ("); + expect(textResult(result)).toContain("go build"); + } finally { + tempDir.removeSync(); + } + }); + it("detects Windows local .exe LSP shims in node_modules/.bin", async () => { if (process.platform !== "win32") { return; From 7fa2c3f42ddce1609f49f4f9ba66ae3254193e7f Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 07:22:28 +0000 Subject: [PATCH 046/205] fix(coding-agent): preserved fork prompt cache affinity - Persisted an inherited provider prompt-cache key on full session forks while keeping the child OMP session id independent. - Added --prompt-cache-key and SDK startup inheritance so explicit cache affinity is separate from provider session routing. - Cleared automatic inherited keys when model, thinking, system prompt, or tool schema inputs change. Fixes #5035 --- ...ion-operations-export-share-fork-resume.md | 6 +- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/cli/args.ts | 1 + packages/coding-agent/src/cli/flag-tables.ts | 3 + packages/coding-agent/src/main.ts | 17 ++ packages/coding-agent/src/sdk.ts | 24 ++- .../coding-agent/src/session/agent-session.ts | 46 +++++ .../src/session/session-entries.ts | 4 + .../src/session/session-manager.ts | 10 +- .../session-fork-prompt-cache-key.test.ts | 194 ++++++++++++++++++ 10 files changed, 306 insertions(+), 3 deletions(-) create mode 100644 packages/coding-agent/test/session-fork-prompt-cache-key.test.ts diff --git a/docs/session-operations-export-share-fork-resume.md b/docs/session-operations-export-share-fork-resume.md index 417827b1a..010512b69 100644 --- a/docs/session-operations-export-share-fork-resume.md +++ b/docs/session-operations-export-share-fork-resume.md @@ -172,7 +172,7 @@ Interactive `/fork` creates a new session from the current one and switches the 2. Flushes pending writes. 3. Calls `SessionManager.fork()`. 4. Copies artifacts directory from old session namespace to new namespace (best-effort; non-ENOENT copy failures are logged, not fatal). -5. Updates `agent.sessionId`. +5. Updates `agent.sessionId` and inherits the previous provider prompt-cache key unless an explicit prompt-cache key is already pinned. 6. Emits `session_switch` with `reason: "fork"`. `SessionManager.fork()` behavior: @@ -184,6 +184,7 @@ Interactive `/fork` creates a new session from the current one and switches the - new timestamp - `cwd` unchanged - `parentSession` set to previous session id + - `providerPromptCacheKey` set to the previous header's inherited key, or the previous session id when none was pinned - Keeps all non-header entries unchanged in the new file. ### Non-persistent behavior @@ -200,6 +201,9 @@ Startup `--fork` is resolved before normal session creation: 2. Path-like values (`/`, `\`, or `.jsonl`) call `SessionManager.forkFrom(path, cwd, sessionDir)`. 3. Other values resolve via `resolveResumableSession(...)`: local sessions first, then global search when `sessionDir` is not forced. Matching accepts lowercased session id prefixes, full JSONL filename prefixes, and timestamp-stripped filename id suffixes. 4. The forked file is created in the current cwd/session-dir scope and becomes the active session manager for startup. +5. Full-context forks automatically seed `providerPromptCacheKey` from the source header's inherited key, falling back to the source session id. Startup drops that automatic inheritance when `--model`, `--thinking`, `--system-prompt`, `--append-system-prompt`, `--tools`, or `--no-tools` changes the provider route or prompt/tool shape. + +Use `--prompt-cache-key ` to pin the provider prompt-cache identity explicitly and independently from both the OMP session id and `--provider-session-id`. `--provider-session-id` continues to control provider session/routing headers and sticky credential selection; `--prompt-cache-key` controls the OpenAI Responses `prompt_cache_key` payload where supported. ## Resume and continue diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..c00e04586 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed full-context forks cold-missing OpenAI prompt caches by persisting an inherited provider prompt-cache key separately from the new OMP session id, adding `--prompt-cache-key` for explicit cache affinity, and dropping automatic inheritance when startup changes the model, thinking level, system prompt, or tool schema. ([#5035](https://github.com/can1357/oh-my-pi/issues/5035)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 8ab71f1f8..d5a0b18d4 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -42,6 +42,7 @@ export interface Args { noSession?: boolean; sessionDir?: string; providerSessionId?: string; + providerPromptCacheKey?: string; fork?: string; /** Collab link to join at startup (set by the `join` subcommand; no CLI flag). */ join?: string; diff --git a/packages/coding-agent/src/cli/flag-tables.ts b/packages/coding-agent/src/cli/flag-tables.ts index e43fc5445..e8c8c9570 100644 --- a/packages/coding-agent/src/cli/flag-tables.ts +++ b/packages/coding-agent/src/cli/flag-tables.ts @@ -141,6 +141,9 @@ export const STRING_SETTERS: Record = { "--provider-session-id": (result, value) => { result.providerSessionId = value; }, + "--prompt-cache-key": (result, value) => { + result.providerPromptCacheKey = value; + }, "--session-dir": (result, value) => { result.sessionDir = value; }, diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index e832c23f2..954c206ec 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -820,6 +820,23 @@ async function buildSessionOptions( if (parsed.providerSessionId) { options.providerSessionId = parsed.providerSessionId; } + if (parsed.providerPromptCacheKey) { + options.providerPromptCacheKey = parsed.providerPromptCacheKey; + options.providerPromptCacheKeySource = "explicit"; + } else { + const header = sessionManager?.getHeader(); + const forkCacheShapeChanged = + parsed.model !== undefined || + parsed.thinking !== undefined || + parsed.systemPrompt !== undefined || + parsed.appendSystemPrompt !== undefined || + parsed.tools !== undefined || + parsed.noTools === true; + if (!forkCacheShapeChanged && header?.providerPromptCacheKey) { + options.providerPromptCacheKey = header.providerPromptCacheKey; + options.providerPromptCacheKeySource = "fork"; + } + } // Model from CLI // - supports --provider --model diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0d27eac07..a87e637c0 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -426,6 +426,8 @@ export interface CreateAgentSessionOptions { providerSessionId?: string; /** Optional provider-facing prompt cache key, distinct from request lineage. */ providerPromptCacheKey?: string; + /** Whether `providerPromptCacheKey` is caller-pinned or inherited from a full fork. */ + providerPromptCacheKeySource?: "explicit" | "fork"; /** Absolute wall-clock deadline in Unix epoch milliseconds. */ deadline?: number; @@ -1221,6 +1223,25 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} SessionManager.create(cwd, SessionManager.getDefaultSessionDir(cwd, agentDir)), ); const providerSessionId = options.providerSessionId ?? sessionManager.getSessionId(); + const forkCacheShapeChanged = + options.model !== undefined || + options.modelPattern !== undefined || + options.thinkingLevel !== undefined || + options.systemPrompt !== undefined || + options.customSystemPrompt !== undefined || + options.appendSystemPrompt !== undefined || + options.toolNames !== undefined || + options.customTools !== undefined; + const inheritedPromptCacheKey = forkCacheShapeChanged + ? undefined + : sessionManager.getHeader()?.providerPromptCacheKey; + const providerPromptCacheKey = options.providerPromptCacheKey ?? inheritedPromptCacheKey; + const providerPromptCacheKeySource = + options.providerPromptCacheKey !== undefined + ? (options.providerPromptCacheKeySource ?? "explicit") + : providerPromptCacheKey !== undefined + ? "fork" + : undefined; // Startup model *selection* only needs to know whether auth is configured for // a candidate's provider — never the resolved key bytes. Use the synchronous, // side-effect-free probe (`hasConfiguredAuth`): it refreshes no OAuth tokens, @@ -2704,7 +2725,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} onPayload, onResponse, sessionId: providerSessionId, - promptCacheKey: options.providerPromptCacheKey, + promptCacheKey: providerPromptCacheKey, deadline: options.deadline, transformContext, transformProviderContext, @@ -2888,6 +2909,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} agentId: resolvedAgentId, agentKind, providerSessionId: options.providerSessionId, + providerPromptCacheKeySource, parentEvalSessionId: options.parentEvalSessionId, advisorTools, titleSystemPrompt: options.titleSystemPrompt, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..b90d7663b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -793,6 +793,8 @@ export interface AgentSessionConfig { * so that credential sticky selection is consistent with the session's streaming calls. */ providerSessionId?: string; + /** Marks `agent.promptCacheKey` as fork-inherited so incompatible route changes can clear it. */ + providerPromptCacheKeySource?: "explicit" | "fork"; /** * Full advisor toolset, pre-built in `createAgentSession` against a distinct, * advisor-scoped `ToolSession` (its own `-advisor` session/agent id) so the @@ -1697,6 +1699,7 @@ export class AgentSession { #agentKind: "main" | "sub" = "main"; #providerSessionId: string | undefined; #freshProviderSessionId: string | undefined; + #inheritedProviderPromptCacheKey: string | undefined; #isDisposed = false; // Extension system #extensionRunner: ExtensionRunner | undefined = undefined; @@ -2212,6 +2215,8 @@ export class AgentSession { this.#agentId = config.agentId; this.#agentKind = config.agentKind ?? "main"; this.#providerSessionId = config.providerSessionId; + this.#inheritedProviderPromptCacheKey = + config.providerPromptCacheKeySource === "fork" ? this.agent.promptCacheKey : undefined; this.agent.setAssistantMessageEventInterceptor((message, assistantMessageEvent) => { const event: AgentEvent = { type: "message_update", @@ -5570,6 +5575,23 @@ export class AgentSession { return this.#freshProviderSessionId ?? this.#providerSessionId ?? sessionId ?? this.sessionManager.getSessionId(); } + #adoptInheritedProviderPromptCacheKey(): void { + const key = this.sessionManager.getHeader()?.providerPromptCacheKey; + if (!key) return; + if (this.#inheritedProviderPromptCacheKey !== undefined || this.agent.promptCacheKey === undefined) { + this.agent.promptCacheKey = key; + this.#inheritedProviderPromptCacheKey = key; + } + } + + #clearInheritedProviderPromptCacheKey(): void { + const key = this.#inheritedProviderPromptCacheKey; + this.#inheritedProviderPromptCacheKey = undefined; + if (key !== undefined && this.agent.promptCacheKey === key) { + this.agent.promptCacheKey = undefined; + } + } + /** * Set agent.sessionId from the session manager and install a dynamic * metadata resolver so every Anthropic API request carries @@ -6375,6 +6397,9 @@ export class AgentSession { if (this.#rebuildSystemPrompt) { const signature = this.#computeAppliedToolSignature(validToolNames, tools); if (signature !== this.#lastAppliedToolSignature) { + if (this.#lastAppliedToolSignature !== undefined) { + this.#clearInheritedProviderPromptCacheKey(); + } const built = await this.#rebuildSystemPrompt(validToolNames, this.#toolRegistry); this.#baseSystemPrompt = built.systemPrompt; this.#baseSystemPromptBeforeMemoryPromotion = undefined; @@ -6461,9 +6486,16 @@ export class AgentSession { if (!this.#rebuildSystemPrompt) return; const activeToolNames = this.getActiveToolNames(); this.#setActiveToolNames?.(activeToolNames); + const previousBaseSystemPrompt = this.#baseSystemPrompt; const built = await this.#rebuildSystemPrompt(activeToolNames, this.#toolRegistry); this.#baseSystemPrompt = built.systemPrompt; this.#baseSystemPromptBeforeMemoryPromotion = undefined; + if ( + previousBaseSystemPrompt.length !== this.#baseSystemPrompt.length || + previousBaseSystemPrompt.some((part, index) => part !== this.#baseSystemPrompt[index]) + ) { + this.#clearInheritedProviderPromptCacheKey(); + } this.agent.setSystemPrompt(this.#baseSystemPrompt); this.#promptModelKey = this.#currentPromptModelKey(); // Refresh the cached signature so a subsequent `#applyActiveToolsByName` with @@ -8704,6 +8736,7 @@ export class AgentSession { this.#clearCheckpointRuntimeState(); this.setTodoPhases([]); this.#freshProviderSessionId = undefined; + this.#clearInheritedProviderPromptCacheKey(); this.#syncAgentSessionId(); this.#rekeyHindsightMemoryForCurrentSessionId(); this.#rekeyMnemopiMemoryForCurrentSessionId(); @@ -8804,6 +8837,7 @@ export class AgentSession { // Update agent session ID this.#freshProviderSessionId = undefined; + this.#adoptInheritedProviderPromptCacheKey(); this.#syncAgentSessionId(); this.#rekeyHindsightMemoryForCurrentSessionId(); this.#rekeyMnemopiMemoryForCurrentSessionId(); @@ -9125,6 +9159,9 @@ export class AgentSession { this.#autoThinking = true; this.#autoResolvedLevel = undefined; this.#thinkingLevel = provisional; + if (!wasAuto) { + this.#clearInheritedProviderPromptCacheKey(); + } this.#applyThinkingLevelToAgent(provisional); if (persist) { this.settings.set("defaultThinkingLevel", AUTO_THINKING); @@ -9148,6 +9185,7 @@ export class AgentSession { this.#applyThinkingLevelToAgent(effectiveLevel); if (isChanging) { + this.#clearInheritedProviderPromptCacheKey(); this.sessionManager.appendThinkingLevelChange(effectiveLevel, effectiveLevel); if (persist && effectiveLevel !== undefined && effectiveLevel !== ThinkingLevel.Off) { this.settings.set("defaultThinkingLevel", effectiveLevel); @@ -11538,6 +11576,9 @@ export class AgentSession { const currentModel = this.model; if (currentModel) { this.#closeProviderSessionsForModelSwitch(currentModel, model); + if (!modelsAreEqual(currentModel, model)) { + this.#clearInheritedProviderPromptCacheKey(); + } } this.agent.setModel(model); @@ -14679,6 +14720,7 @@ export class AgentSession { const previousSystemPrompt = this.agent.state.systemPrompt; const previousBaseSystemPromptBeforeMemoryPromotion = this.#baseSystemPromptBeforeMemoryPromotion; const previousFreshProviderSessionId = this.#freshProviderSessionId; + const previousInheritedProviderPromptCacheKey = this.#inheritedProviderPromptCacheKey; const previousFallbackSelectedMCPToolNames = previousSessionFile ? this.#getSessionDefaultSelectedMCPToolNames(previousSessionFile) : undefined; @@ -14700,6 +14742,8 @@ export class AgentSession { await this.sessionManager.setSessionFile(sessionPath); if (switchingToDifferentSession) { this.#freshProviderSessionId = undefined; + this.#clearInheritedProviderPromptCacheKey(); + this.#adoptInheritedProviderPromptCacheKey(); } this.#syncAgentSessionId(); this.#rekeyHindsightMemoryForCurrentSessionId(); @@ -14853,6 +14897,7 @@ export class AgentSession { this.agent.replaceQueues(previousSteeringMessages, previousFollowUpMessages); this.#pendingNextTurnMessages = previousPendingNextTurnMessages; this.#scheduledHiddenNextTurnGeneration = previousScheduledHiddenNextTurnGeneration; + this.#inheritedProviderPromptCacheKey = previousInheritedProviderPromptCacheKey; this.#checkpointState = previousCheckpointState; this.#pendingRewindReport = previousPendingRewindReport; this.#lastCompletedRewind = previousLastCompletedRewind; @@ -14928,6 +14973,7 @@ export class AgentSession { this.#rehydrateCheckpointRewindState(); this.#syncTodoPhasesFromBranch(); this.#freshProviderSessionId = undefined; + this.#clearInheritedProviderPromptCacheKey(); this.#syncAgentSessionId(); this.#rekeyHindsightMemoryForCurrentSessionId(); this.#rekeyMnemopiMemoryForCurrentSessionId(); diff --git a/packages/coding-agent/src/session/session-entries.ts b/packages/coding-agent/src/session/session-entries.ts index bccfb9624..59ed3e506 100644 --- a/packages/coding-agent/src/session/session-entries.ts +++ b/packages/coding-agent/src/session/session-entries.ts @@ -32,10 +32,14 @@ export interface SessionHeader { timestamp: string; cwd: string; parentSession?: string; + /** Provider prompt-cache identity inherited by exact-route full forks. */ + providerPromptCacheKey?: string; } export interface NewSessionOptions { parentSession?: string; + /** Provider prompt-cache identity to seed on the new session header. */ + providerPromptCacheKey?: string; /** Skip flushing the current session and delete it instead of saving. */ drop?: boolean; } diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index 741df118d..1f27a1294 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -783,6 +783,7 @@ export class SessionManager { timestamp, cwd: this.#cwd, parentSession: options?.parentSession, + providerPromptCacheKey: options?.providerPromptCacheKey, }; this.#titleUpdatedAt = timestamp; @@ -1050,6 +1051,7 @@ export class SessionManager { timestamp, cwd: this.#cwd, parentSession: parentSessionId, + providerPromptCacheKey: this.#header.providerPromptCacheKey ?? parentSessionId, }; this.#sessionName = this.#header.title; this.#titleSource = this.#header.titleSource; @@ -1888,7 +1890,13 @@ export class SessionManager { const sourceHeader = sourceEntries.find(entry => entry.type === "session") as SessionHeader | undefined; const history = sourceEntries.filter(entry => entry.type !== "session") as SessionEntry[]; - manager.#resetToNewSession({ parentSession: sourceHeader?.id }, options?.sessionFile); + manager.#resetToNewSession( + { + parentSession: sourceHeader?.id, + providerPromptCacheKey: sourceHeader?.providerPromptCacheKey ?? sourceHeader?.id, + }, + options?.sessionFile, + ); manager.#header.title = sourceHeader?.title; manager.#header.titleSource = sourceHeader?.titleSource; manager.#sessionName = manager.#header.title; diff --git a/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts b/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts new file mode 100644 index 000000000..f16fd84d9 --- /dev/null +++ b/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts @@ -0,0 +1,194 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { type Args, parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { type CreateAgentSessionOptions, createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; +import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { CURRENT_SESSION_VERSION, type SessionHeader } from "@oh-my-pi/pi-coding-agent/session/session-entries"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +const OPENAI_TEST_MODEL = getBundledModel("openai", "gpt-4o-mini"); + +interface ArgsWithPromptCacheKey extends Args { + providerPromptCacheKey?: string; +} + +interface SourceSessionFixture { + cwd: string; + sourceFile: string; + sourceHeader: SessionHeader; + forkSessionDir: string; +} + +async function createSourceSessionFixture(tempDir: TempDir, parentId: string): Promise { + const cwd = tempDir.join("project"); + const sourceDir = tempDir.join("source-sessions"); + const forkSessionDir = tempDir.join("forked-sessions"); + await fs.mkdir(cwd, { recursive: true }); + await fs.mkdir(sourceDir, { recursive: true }); + await fs.mkdir(forkSessionDir, { recursive: true }); + const sourceFile = path.join(sourceDir, `${parentId}.jsonl`); + const sourceHeader: SessionHeader = { + type: "session", + version: CURRENT_SESSION_VERSION, + id: parentId, + timestamp: new Date().toISOString(), + cwd, + }; + await Bun.write(sourceFile, `${JSON.stringify(sourceHeader)}\n`); + return { cwd, sourceFile, sourceHeader, forkSessionDir }; +} + +async function createMinimalSession( + tempDir: TempDir, + options: CreateAgentSessionOptions, +): Promise<{ session: AgentSession; authStorage: AuthStorage }> { + const authStorage = await AuthStorage.create(tempDir.join("sdk-auth.db")); + authStorage.setRuntimeApiKey("openai", "test-key"); + const shouldSupplyModel = options.sessionManager?.getHeader()?.parentSession === undefined; + const result = await createAgentSession({ + ...options, + cwd: options.cwd ?? tempDir.path(), + agentDir: tempDir.path(), + authStorage, + modelRegistry: undefined, + model: shouldSupplyModel ? (options.model ?? OPENAI_TEST_MODEL) : options.model, + settings: Settings.isolated({ + "async.enabled": false, + "marketplace.autoUpdate": "off", + }), + disableExtensionDiscovery: true, + preloadedExtensions: undefined, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + workspaceTree: { + rootPath: options.cwd ?? tempDir.path(), + rendered: "", + truncated: false, + totalLines: 0, + agentsMdFiles: [], + }, + enableMCP: false, + enableLsp: false, + ...(options.toolNames !== undefined ? { toolNames: options.toolNames } : {}), + }); + return { session: result.session, authStorage }; +} + +describe("provider prompt-cache key session affinity", () => { + it("parses --prompt-cache-key without folding it into provider session id or prompt text", () => { + const parsed = parseArgs([ + "--provider-session-id", + "provider-lineage", + "--prompt-cache-key", + "cache-affinity", + "hello", + ]); + const promptCacheArgs: ArgsWithPromptCacheKey = parsed; + + expect(parsed.providerSessionId).toBe("provider-lineage"); + expect(promptCacheArgs.providerPromptCacheKey).toBe("cache-affinity"); + expect(parsed.messages).toEqual(["hello"]); + expect(parsed.unrecognizedFlags).toEqual([]); + }); + + it("creates an agent whose prompt-cache key can differ from provider request lineage", async () => { + using tempDir = TempDir.createSync("@omp-prompt-cache-sdk-"); + let session: AgentSession | undefined; + let authStorage: AuthStorage | undefined; + try { + const created = await createMinimalSession(tempDir, { + providerSessionId: "provider-lineage", + providerPromptCacheKey: "cache-affinity", + sessionManager: SessionManager.inMemory(tempDir.path()), + }); + session = created.session; + authStorage = created.authStorage; + + expect(session.agent.sessionId).toBe("provider-lineage"); + expect(session.agent.promptCacheKey).toBe("cache-affinity"); + expect(session.agent.promptCacheKey).not.toBe(session.agent.sessionId); + } finally { + await session?.dispose(); + authStorage?.close(); + } + }); + + it("initializes a full fork with child request lineage and parent prompt-cache affinity", async () => { + using tempDir = TempDir.createSync("@omp-prompt-cache-fork-"); + const source = await createSourceSessionFixture(tempDir, "parent-cache-session"); + const forkedManager = await SessionManager.forkFrom(source.sourceFile, source.cwd, source.forkSessionDir); + let session: AgentSession | undefined; + let authStorage: AuthStorage | undefined; + try { + const created = await createMinimalSession(tempDir, { + cwd: source.cwd, + sessionManager: forkedManager, + }); + session = created.session; + authStorage = created.authStorage; + const childSessionId = forkedManager.getSessionId(); + + expect(forkedManager.getHeader()?.parentSession).toBe(source.sourceHeader.id); + expect(childSessionId).toBeString(); + expect(childSessionId).not.toBe(source.sourceHeader.id); + expect(session.agent.sessionId).toBe(childSessionId); + expect(session.agent.promptCacheKey).toBe(source.sourceHeader.id); + expect(session.agent.promptCacheKey).not.toBe(session.agent.sessionId); + } finally { + await session?.dispose(); + authStorage?.close(); + } + }); + + it("does not auto-inherit parent prompt-cache affinity when fork startup changes request-shaping inputs", async () => { + const cases: Array<{ name: string; options: CreateAgentSessionOptions }> = [ + { + name: "model", + options: { model: OPENAI_TEST_MODEL }, + }, + { + name: "thinking", + options: { thinkingLevel: ThinkingLevel.High }, + }, + { + name: "system", + options: { customSystemPrompt: "Use a different provider prompt." }, + }, + { + name: "tools", + options: { toolNames: ["read"] }, + }, + ]; + + for (const entry of cases) { + using tempDir = TempDir.createSync(`@omp-prompt-cache-fork-${entry.name}-`); + const source = await createSourceSessionFixture(tempDir, `parent-cache-session-${entry.name}`); + const forkedManager = await SessionManager.forkFrom(source.sourceFile, source.cwd, source.forkSessionDir); + let session: AgentSession | undefined; + let authStorage: AuthStorage | undefined; + try { + const created = await createMinimalSession(tempDir, { + ...entry.options, + cwd: source.cwd, + sessionManager: forkedManager, + }); + session = created.session; + authStorage = created.authStorage; + + expect(forkedManager.getHeader()?.parentSession).toBe(source.sourceHeader.id); + expect(session.agent.promptCacheKey, entry.name).toBeUndefined(); + } finally { + await session?.dispose(); + authStorage?.close(); + } + } + }); +}); From 325375f801d72c4119837a37bd50ca56160ee1ac Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 07:24:36 +0000 Subject: [PATCH 047/205] fix(advisor): used uuidv7 codex session ids Separated advisor provider session identity from local advisor labels so Codex requests carry stable UUIDv7 values while transcripts keep their advisor-specific names. Fixes #5040 --- packages/coding-agent/CHANGELOG.md | 4 + .../src/advisor/__tests__/config.test.ts | 84 +++++++++++++++++++ packages/coding-agent/src/advisor/config.ts | 28 +++++++ .../coding-agent/src/session/agent-session.ts | 72 +++++++++------- 4 files changed, 158 insertions(+), 30 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..34d5518aa 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Codex advisor requests using local `-advisor` session labels as provider session IDs; advisors now use stable UUIDv7 provider identities while keeping labeled transcript names. ([#5040](https://github.com/can1357/oh-my-pi/issues/5040)) + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/advisor/__tests__/config.test.ts b/packages/coding-agent/src/advisor/__tests__/config.test.ts index 3c554e168..44d14f936 100644 --- a/packages/coding-agent/src/advisor/__tests__/config.test.ts +++ b/packages/coding-agent/src/advisor/__tests__/config.test.ts @@ -5,6 +5,7 @@ import * as path from "node:path"; import { advisorConfigFilePath, discoverAdvisorConfigs, + getOrCreateAdvisorProviderSessionId, loadWatchdogConfigFile, resolveAdvisorConfigEditPath, saveWatchdogConfigFile, @@ -86,6 +87,89 @@ describe("slugifyAdvisorName", () => { }); }); +describe("getOrCreateAdvisorProviderSessionId", () => { + const primarySessionA = "018f8f5d-75b0-7cc6-8a6f-2f1c0b8e4c9d"; + const primarySessionB = "018f8f5d-75b1-7cc6-8a6f-2f1c0b8e4c9d"; + + it("returns the generated UUIDv7 instead of a local advisor label", () => { + const generated = "0193c8f2-7b1a-7c4d-9e2f-123456789abc"; + + const providerSessionId = getOrCreateAdvisorProviderSessionId( + new Map(), + primarySessionA, + "security-advisor", + () => generated, + ); + + expect(providerSessionId).toBe(generated); + expect(providerSessionId).not.toContain("-advisor"); + }); + + it("reuses the same generated UUIDv7 for repeated calls with the same primary session and slug", () => { + const generatedIds = ["0193c8f2-7b1a-7c4d-9e2f-123456789abc", "0193c8f2-7b1b-7c4d-9e2f-123456789abc"]; + let nextGeneratedIdIndex = 0; + const ids = new Map(); + + const first = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "architecture", () => { + const generated = generatedIds[nextGeneratedIdIndex]; + if (!generated) throw new Error("unexpected generator call"); + nextGeneratedIdIndex += 1; + return generated; + }); + const second = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "architecture", () => { + const generated = generatedIds[nextGeneratedIdIndex]; + if (!generated) throw new Error("unexpected generator call"); + nextGeneratedIdIndex += 1; + return generated; + }); + + expect(first).toBe(generatedIds[0]); + expect(second).toBe(generatedIds[0]); + expect(nextGeneratedIdIndex).toBe(1); + }); + + it("creates distinct UUIDv7 values for different advisor slugs or primary sessions", () => { + const generatedIds = [ + "0193c8f2-7b1a-7c4d-9e2f-123456789abc", + "0193c8f2-7b1b-7c4d-9e2f-123456789abc", + "0193c8f2-7b1c-7c4d-9e2f-123456789abc", + ]; + let nextGeneratedIdIndex = 0; + const ids = new Map(); + const nextGeneratedId = () => { + const generated = generatedIds[nextGeneratedIdIndex]; + if (!generated) throw new Error("unexpected generator call"); + nextGeneratedIdIndex += 1; + return generated; + }; + + const architecture = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "architecture", nextGeneratedId); + const security = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "security", nextGeneratedId); + const architectureForOtherSession = getOrCreateAdvisorProviderSessionId( + ids, + primarySessionB, + "architecture", + nextGeneratedId, + ); + + expect(architecture).toBe(generatedIds[0]); + expect(security).toBe(generatedIds[1]); + expect(architectureForOtherSession).toBe(generatedIds[2]); + expect(new Set([architecture, security, architectureForOtherSession]).size).toBe(3); + }); + + it("rejects generated values that are not UUIDv7", () => { + expect(() => + getOrCreateAdvisorProviderSessionId( + new Map(), + primarySessionA, + "architecture", + () => "550e8400-e29b-41d4-a716-446655440000", + ), + ).toThrow("non-UUIDv7"); + }); +}); + describe("WATCHDOG.yml file round-trip", () => { let tmp: string; beforeEach(async () => { diff --git a/packages/coding-agent/src/advisor/config.ts b/packages/coding-agent/src/advisor/config.ts index 0ee5bbe8d..1d0f15bb5 100644 --- a/packages/coding-agent/src/advisor/config.ts +++ b/packages/coding-agent/src/advisor/config.ts @@ -59,6 +59,34 @@ export function slugifyAdvisorName(name: string): string { return slug || "advisor"; } +const UUID_V7_PATTERN = /^[0-9a-f]{8}-[0-9a-f]{4}-7[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i; +const ADVISOR_PROVIDER_SESSION_KEY_SEPARATOR = "\u0000"; + +/** + * Returns a stable provider-facing UUIDv7 for one advisor within one primary session. + * + * Codex treats `session_id`/`conversation_id` as a UUID-shaped routing identity, + * so advisor labels such as `-advisor` stay local-only. + */ +export function getOrCreateAdvisorProviderSessionId( + ids: Map, + primarySessionId: string | undefined, + slug: string, + randomSessionId: () => string = () => Bun.randomUUIDv7(), +): string | undefined { + if (!primarySessionId) return undefined; + const key = `${primarySessionId}${ADVISOR_PROVIDER_SESSION_KEY_SEPARATOR}${slug}`; + const existing = ids.get(key); + if (existing) return existing; + + const next = randomSessionId(); + if (!UUID_V7_PATTERN.test(next)) { + throw new Error("Advisor provider session id generator returned a non-UUIDv7 value"); + } + ids.set(key, next); + return next; +} + /** Built tool names, for validating an advisor's `tools` list. */ const KNOWN_TOOL_NAMES = new Set(BUILTIN_TOOL_NAMES); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..44b1d605f 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -154,6 +154,7 @@ import { AdvisorTranscriptRecorder, advisorTranscriptFilename, formatAdvisorBatchContent, + getOrCreateAdvisorProviderSessionId, isAdvisorInterruptImmuneTurnActive, isInterruptingSeverity, resolveAdvisorDeliveryChannel, @@ -1599,6 +1600,8 @@ export class AgentSession { #advisors: ActiveAdvisor[] = []; /** Configured advisor roster from WATCHDOG.yml; undefined/empty → single legacy advisor. */ #advisorConfigs?: AdvisorConfig[]; + /** Provider-facing UUIDv7 identities keyed by primary provider session and advisor slug. */ + #advisorProviderSessionIds = new Map(); /** Aggregate of the most recent stop's recorder closes; awaited by dispose() and * used as the open barrier for the next build so two writers never share a file. */ #advisorRecorderClosed: Promise = Promise.resolve(); @@ -2477,18 +2480,26 @@ export class AgentSession { const names = config.tools?.length ? new Set(config.tools) : ADVISOR_DEFAULT_TOOL_NAMES; const tools = (this.#advisorTools ?? []).filter(t => names.has(t.name)); - const advisorSessionId = this.#advisorSessionId(slug); + const primaryProviderSessionId = this.sessionId; + const advisorSessionLabel = slug + ? `${primaryProviderSessionId}-advisor-${slug}` + : `${primaryProviderSessionId}-advisor`; + const advisorProviderSessionId = getOrCreateAdvisorProviderSessionId( + this.#advisorProviderSessionIds, + primaryProviderSessionId, + slug, + ); const appendOnlyContext = new AppendOnlyContextManager(); // Thread the primary's telemetry into the advisor loop so the advisor - // model's GenAI spans + usage/cost hooks fire stamped with the advisor's - // own identity. `conversationId` is cleared so the advisor loop falls back - // to its own session id; undefined telemetry stays undefined. + // model's GenAI spans + usage/cost hooks fire stamped with the local advisor + // identity. `conversationId` is cleared so provider telemetry falls back to + // the UUIDv7 provider session id, not the local `-advisor` label. const advisorTelemetry = this.agent.telemetry ? { ...this.agent.telemetry, agent: { - id: advisorSessionId, + id: advisorSessionLabel, name: slug ? `${MODEL_ROLES.advisor.name}: ${advisorName}` : MODEL_ROLES.advisor.name, description: formatModelString(advisorModel), }, @@ -2500,10 +2511,10 @@ export class AgentSession { // advisor's requests cache, route, and obfuscate like the main turn. // `promptCacheKey` preserves an explicitly pinned provider cache key // unchanged so tan/shared-session advisor calls read the exact shard the - // parent turn populated, while keeping only `sessionId` advisor-scoped; - // sessions without a pinned key fall back to the advisor session id for - // stable advisor-local caching (see can1357/oh-my-pi#3639). - const advisorPromptCacheKey = this.agent.promptCacheKey ?? advisorSessionId; + // parent turn populated. Otherwise the advisor uses its provider UUIDv7 so + // Codex request identity remains UUID-shaped while local labels keep the + // `-advisor` suffix. + const advisorPromptCacheKey = this.agent.promptCacheKey ?? advisorProviderSessionId; const advisorAgent = new Agent({ initialState: { systemPrompt, @@ -2512,11 +2523,11 @@ export class AgentSession { tools: [adviseTool, ...tools], }, appendOnlyContext, - sessionId: advisorSessionId, + sessionId: advisorProviderSessionId, promptCacheKey: advisorPromptCacheKey, providerSessionState: this.#providerSessionState, preferWebsockets: this.#preferWebsockets, - getApiKey: requestModel => this.#modelRegistry.resolver(requestModel, advisorSessionId), + getApiKey: requestModel => this.#modelRegistry.resolver(requestModel, advisorProviderSessionId), streamFn: this.#advisorStreamFn, onPayload: this.#onPayload, onResponse: this.#onResponse, @@ -2575,11 +2586,15 @@ export class AgentSession { // suspect-mark a credential on a transient advisor error). const message = error instanceof Error ? error.message : String(error); if (!isUsageLimitOutcome(extractHttpStatusFromError(error), message)) return; - await this.#modelRegistry.authStorage.markUsageLimitReached(advisorModel.provider, advisorSessionId, { - retryAfterMs: extractRetryHint(undefined, message), - baseUrl: advisorModel.baseUrl, - modelId: advisorModel.id, - }); + await this.#modelRegistry.authStorage.markUsageLimitReached( + advisorModel.provider, + advisorProviderSessionId, + { + retryAfterMs: extractRetryHint(undefined, message), + baseUrl: advisorModel.baseUrl, + modelId: advisorModel.id, + }, + ); }, notifyFailure: error => { const message = error instanceof Error ? error.message : String(error); @@ -2633,13 +2648,6 @@ export class AgentSession { return this.#advisors.length > 0; } - /** Provider/session id for an advisor's loop. The slug suffix MUST match the - * advisor's transcript filename so stats/telemetry attribute the same advisor. */ - #advisorSessionId(slug: string): string | undefined { - if (!this.sessionId) return undefined; - return slug ? `${this.sessionId}-advisor-${slug}` : `${this.sessionId}-advisor`; - } - /** * Route one accepted advice note from `advisor` to the primary. Concern and * blocker interrupt the running agent through the steering channel; once the @@ -2845,11 +2853,15 @@ export class AgentSession { // No compaction candidates, fallback to re-prime return true; } - const advisorSessionId = this.#advisorSessionId(advisor.slug); + const advisorProviderSessionId = getOrCreateAdvisorProviderSessionId( + this.#advisorProviderSessionIds, + this.sessionId, + advisor.slug, + ); const preparation = prepareCompaction( pathEntries, compactionSettings, - await this.#runnableCompactionCandidates(candidates, advisorSessionId), + await this.#runnableCompactionCandidates(candidates, advisorProviderSessionId), ); if (!preparation) { // Cannot prepare compaction, fallback to re-prime @@ -2869,17 +2881,17 @@ export class AgentSession { let lastError: unknown; // Instrument the advisor's overflow-compaction one-shot like the primary // compaction path so the advisor model's maintenance call also emits spans. - const telemetry = resolveTelemetry(agent.telemetry, advisorSessionId); + const telemetry = resolveTelemetry(agent.telemetry, advisorProviderSessionId); for (const candidate of candidates) { - const apiKey = await this.#modelRegistry.getApiKey(candidate, advisorSessionId); + const apiKey = await this.#modelRegistry.getApiKey(candidate, advisorProviderSessionId); if (!apiKey) continue; try { compactResult = await compact( preparation, candidate, - this.#modelRegistry.resolver(candidate, advisorSessionId), + this.#modelRegistry.resolver(candidate, advisorProviderSessionId), undefined, undefined, { @@ -2887,8 +2899,8 @@ export class AgentSession { convertToLlm: messages => this.#convertToLlmForSideRequest(messages), telemetry, tools: agent.state.tools, - sessionId: advisorSessionId, - promptCacheKey: advisorSessionId, + sessionId: advisorProviderSessionId, + promptCacheKey: advisorProviderSessionId, }, ); break; From b2b1708d88802feb2fcd73d68f5f35bb32aa1a15 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 07:41:59 +0000 Subject: [PATCH 048/205] fix(coding-agent): guarded fork cache on scoped models - Treated startup scoped model selection as a prompt-cache shape override before inheriting fork cache keys. - Covered the --models fork path so a scoped startup model cannot reuse the parent prompt_cache_key. Fixes #5035 --- packages/coding-agent/src/main.ts | 5 ++- .../session-fork-prompt-cache-key.test.ts | 39 +++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index 954c206ec..ee25259b0 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -789,7 +789,8 @@ export function applyResolvedSystemPromptInputs( } } -async function buildSessionOptions( +/** Builds startup session options from parsed CLI flags, scoped models, and resolved session lineage. */ +export async function buildSessionOptions( parsed: Args, scopedModels: ScopedModel[], sessionManager: SessionManager | undefined, @@ -825,7 +826,9 @@ async function buildSessionOptions( options.providerPromptCacheKeySource = "explicit"; } else { const header = sessionManager?.getHeader(); + const scopedModelOverride = scopedModels.length > 0 && !parsed.continue && !parsed.resume; const forkCacheShapeChanged = + scopedModelOverride || parsed.model !== undefined || parsed.thinking !== undefined || parsed.systemPrompt !== undefined || diff --git a/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts b/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts index f16fd84d9..1c64a15e4 100644 --- a/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts +++ b/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts @@ -4,7 +4,10 @@ import * as path from "node:path"; import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { type Args, parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import type { ScopedModel } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { buildSessionOptions } from "@oh-my-pi/pi-coding-agent/main"; import { type CreateAgentSessionOptions, createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; @@ -191,4 +194,40 @@ describe("provider prompt-cache key session affinity", () => { } } }); + + it("does not pre-pin parent prompt-cache affinity when a scoped model selects the startup route", async () => { + using tempDir = TempDir.createSync("@omp-prompt-cache-scoped-model-"); + const source = await createSourceSessionFixture(tempDir, "parent-cache-session-scoped"); + const forkedManager = await SessionManager.forkFrom(source.sourceFile, source.cwd, source.forkSessionDir); + const authStorage = await AuthStorage.create(tempDir.join("scoped-auth.db")); + authStorage.setRuntimeApiKey(OPENAI_TEST_MODEL.provider, "test-key"); + try { + const modelRegistry = new ModelRegistry(authStorage, tempDir.join("models.yml")); + const parsed = parseArgs([ + "--cwd", + source.cwd, + "--models", + `${OPENAI_TEST_MODEL.provider}/${OPENAI_TEST_MODEL.id}`, + ]); + const scopedModels: ScopedModel[] = [ + { + model: OPENAI_TEST_MODEL, + explicitThinkingLevel: false, + }, + ]; + + const options = await buildSessionOptions( + parsed, + scopedModels, + forkedManager, + modelRegistry, + Settings.isolated({ "marketplace.autoUpdate": "off" }), + ); + + expect(options.model).toBe(OPENAI_TEST_MODEL); + expect(options.providerPromptCacheKey).toBeUndefined(); + } finally { + authStorage.close(); + } + }); }); From 29deeef8763dbb3a14af75fefd8c9677808e1138 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 09:46:27 +0200 Subject: [PATCH 049/205] feat: enabled codex responses lite for gpt-5.6 models and remote compaction - Enabled Codex Responses Lite for GPT-5.6 models by integrating model discovery flags and wire contract updates. - Implemented request transformations for streaming and remote compaction, including header injection and image detail stripping. - Introduced sequential-cutoff logic and atomic reasoning summary events for concurrent stream processing. - Added comprehensive test suites to validate remote compaction, image handling, and reasoning summary delivery. --- packages/agent/CHANGELOG.md | 4 + .../src/compaction/compaction-v2-streaming.ts | 18 +- packages/agent/src/compaction/openai.ts | 8 + packages/agent/test/remote-compaction.test.ts | 99 +++++ packages/ai/CHANGELOG.md | 15 + .../src/providers/openai-codex-responses.ts | 65 ++- .../openai-codex/request-transformer.ts | 119 +++-- packages/ai/src/providers/openai-shared.ts | 28 ++ packages/ai/test/helpers/index.ts | 7 +- .../test/openai-codex-responses-lite.test.ts | 223 +++++++++- packages/catalog/CHANGELOG.md | 13 + packages/catalog/src/discovery/codex.ts | 3 + packages/catalog/src/models.json | 414 +++++++++++++++--- packages/catalog/src/types.ts | 2 + packages/catalog/src/wire/codex.ts | 2 + packages/catalog/test/codex-discovery.test.ts | 44 ++ 16 files changed, 945 insertions(+), 119 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 04b655eeb..cf30037fd 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed remote compaction for Codex Responses Lite models (GPT-5.6 family): both the V1 `/responses/compact` request and the V2 `compaction_trigger` stream now apply the lite rewrite (instructions as an input item, no top-level `instructions`/`tools`, `all_turns` reasoning replay on V2) and send the `x-openai-internal-codex-responses-lite` header, matching codex-rs routing compaction through `build_responses_request`. + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/agent/src/compaction/compaction-v2-streaming.ts b/packages/agent/src/compaction/compaction-v2-streaming.ts index 92ecfcdb0..251573775 100644 --- a/packages/agent/src/compaction/compaction-v2-streaming.ts +++ b/packages/agent/src/compaction/compaction-v2-streaming.ts @@ -9,6 +9,7 @@ import type { Api, FetchImpl, Model } from "@oh-my-pi/pi-ai"; import { isTransientStatus, ProviderHttpError } from "@oh-my-pi/pi-ai/error"; +import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; import { getOpenAIPromptCacheKey, getOpenAIResponsesRoutingSessionId, @@ -276,10 +277,22 @@ async function attemptCompactionV2Streaming( instructions: request.instructions, stream: true, store: false, - ...(request.reasoning ? { reasoning: request.reasoning, include: ["reasoning.encrypted_content"] } : {}), + ...(request.reasoning + ? { + // Lite implies gpt-5.4+, where codex-rs sends `all_turns` replay. + reasoning: model.useResponsesLite ? { ...request.reasoning, context: "all_turns" } : request.reasoning, + include: ["reasoning.encrypted_content"], + } + : {}), ...(promptCacheKey ? { prompt_cache_key: promptCacheKey } : {}), ...(request.tools && request.tools.length > 0 ? { tools: request.tools, tool_choice: "auto" } : {}), }; + // Responses Lite models take the same rewrite on the compaction stream: + // instructions/tools ride as input items (codex-rs `compact_remote_v2` + // builds through `build_responses_request`). + if (model.useResponsesLite) { + applyCodexResponsesLiteShape(body); + } const response = await fetchImpl(endpoint, { method: "POST", headers: buildCompactionV2Headers(model, apiKey, request), @@ -338,6 +351,9 @@ function buildCompactionV2Headers(model: Model, apiKey: string, request: Compact } headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES; headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX; + if (model.useResponsesLite) { + headers[OPENAI_HEADERS.RESPONSES_LITE] = "true"; + } } return headers; diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index 2695e4466..aa4f37bb4 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -16,6 +16,7 @@ */ import { ProviderHttpError } from "@oh-my-pi/pi-ai/error"; +import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; import { parseAzureDeploymentNameMap, parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; import type { Api, AssistantMessage, FetchImpl, Message, Model } from "@oh-my-pi/pi-ai/types"; @@ -493,6 +494,13 @@ export async function requestOpenAiRemoteCompaction( } headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES; headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX; + // Responses Lite models take the same rewrite on `/responses/compact`: + // instructions ride as an input item and the lite marker header is set + // (codex-rs routes compaction through `build_responses_request`). + if (model.useResponsesLite) { + applyCodexResponsesLiteShape(request); + headers[OPENAI_HEADERS.RESPONSES_LITE] = "true"; + } } const response = await (opts?.fetch ?? fetch)(endpoint, { diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index c2cc50dd1..237c81557 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -393,6 +393,105 @@ describe("requestCompactionV2Streaming", () => { }); }); +describe("Responses Lite remote compaction", () => { + function makeCodexLiteModel(): Model<"openai-codex-responses"> { + return buildModel({ + id: "gpt-5.6-terra", + name: "GPT-5.6 Terra", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.example/backend-api", + reasoning: true, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 372000, + maxTokens: 128000, + useResponsesLite: true, + remoteCompaction: { enabled: true, api: "openai-codex-responses", v2StreamingEnabled: true }, + }); + } + + interface CapturedLiteRequest { + instructions?: unknown; + tools?: unknown; + input?: Array>; + } + + function captureLite(init: RequestInit | undefined): { body: CapturedLiteRequest; liteHeader?: string } { + if (!init?.headers || init.headers instanceof Headers || Array.isArray(init.headers)) { + throw new Error("Expected remote compaction to send headers as a plain object"); + } + const rawLite = init.headers["x-openai-internal-codex-responses-lite"]; + return { + body: JSON.parse(String(init.body)) as CapturedLiteRequest, + liteHeader: typeof rawLite === "string" ? rawLite : undefined, + }; + } + + test("V1 compaction sends the lite header and input-item instructions", async () => { + const model = makeCodexLiteModel(); + let captured: { body: CapturedLiteRequest; liteHeader?: string } | undefined; + const fetchMock: FetchImpl = async (_input, init) => { + captured = captureLite(init); + return Response.json({ output: [{ type: "compaction", encrypted_content: "enc" }] }); + }; + + await requestOpenAiRemoteCompaction( + model, + "test-key", + [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + "compact instructions", + undefined, + { fetch: fetchMock }, + ); + + expect(captured?.liteHeader).toBe("true"); + expect(captured?.body.instructions).toBeUndefined(); + expect(captured?.body.tools).toBeUndefined(); + expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); + expect(captured?.body.input?.[1]).toEqual({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: "compact instructions" }], + }); + }); + + test("V2 streaming compaction applies the lite rewrite and keeps the trigger last", async () => { + const model = makeCodexLiteModel(); + const request = buildCompactionV2Request( + model, + [{ type: "message", role: "user", content: [{ type: "input_text", text: "real user" }] }], + "compact instructions", + ); + let captured: { body: CapturedLiteRequest; liteHeader?: string } | undefined; + const fetchMock: FetchImpl = async (_input, init) => { + captured = captureLite(init); + return sseResponse([ + { + type: "response.output_item.done", + output_index: 0, + item: { type: "compaction", encrypted_content: "enc" }, + }, + { type: "response.completed", response: { usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2 } } }, + ]); + }; + + expect(shouldUseCompactionV2Streaming(model)).toBe(true); + await requestCompactionV2Streaming(model, "test-key", request, undefined, { fetch: fetchMock }); + + expect(captured?.liteHeader).toBe("true"); + expect(captured?.body.instructions).toBeUndefined(); + expect(captured?.body.tools).toBeUndefined(); + expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); + expect(captured?.body.input?.[1]).toEqual({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: "compact instructions" }], + }); + expect(captured?.body.input?.at(-1)).toEqual({ type: "compaction_trigger" }); + }); +}); + test("uses configured OpenAI-compatible compaction for custom providers", async () => { const model = makeOpenAiModel({ provider: "cliproxy-codex", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index db588e153..cb7865ac7 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,9 +2,24 @@ ## [Unreleased] +### Added + +- Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. +- Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`. +- Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress. + ### Changed +- Refactored Responses Lite transport to move tools and instructions into input items +- Updated Responses Lite to force parallel tool calling off and strip image detail +- Standardized Responses Lite activation via model-level catalog flags + - Recognized Pro Lite as a paid plan tier for OpenAI Codex models +- Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape. + +### Fixed + +- Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract ## [16.3.15] - 2026-07-09 diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 38bb777a2..86ca0085c 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -65,7 +65,7 @@ import { type CodexRequestOptions, type InputItem, type RequestBody, - shouldUseCodexResponsesLite, + resolveCodexResponsesLite, transformRequestBody, } from "./openai-codex/request-transformer"; import { CodexApiError } from "./openai-codex/response-handler"; @@ -88,6 +88,7 @@ import { appendReasoningSummaryTextDelta, appendResponsesToolResultMessages, applyOpenAIServiceTier, + applyReasoningSummaryDone, buildResponsesDeltaInput, convertResponsesAssistantMessage, convertResponsesInputContent, @@ -117,11 +118,13 @@ export interface OpenAICodexResponsesOptions extends StreamOptions { preferWebsockets?: boolean; serviceTier?: ServiceTier; /** - * Opt into the Responses Lite transport contract. Sends + * Responses Lite transport override; defaults to the model's catalog + * `useResponsesLite` flag (codex-rs `use_responses_lite`). Sends * `x-openai-internal-codex-responses-lite: true` on HTTP requests and on the * WebSocket upgrade (the marker is connection-scoped there, so lite and - * non-lite turns never share a pooled socket), strips image detail from - * input, and disables parallel tool calling — mirroring codex-rs. + * non-lite turns never share a pooled socket), moves instructions/tools + * into input items, strips image detail, and disables parallel tool + * calling — mirroring codex-rs. */ responsesLite?: boolean; /** @@ -186,7 +189,6 @@ const CODEX_RETRYABLE_EVENT_MESSAGE = const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses"; const X_CODEX_TURN_STATE_HEADER = "x-codex-turn-state"; const X_MODELS_ETAG_HEADER = "x-models-etag"; -const X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER = "x-openai-internal-codex-responses-lite"; /** WebSocket frames cannot carry per-request HTTP headers; codex-rs mirrors the lite marker into `client_metadata` under this key. */ const CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY = "ws_request_header_x_openai_internal_codex_responses_lite"; /** `response.metadata` payload key carrying ChatGPT moderation metadata. */ @@ -913,7 +915,7 @@ async function buildCodexRequestContext( }; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); - const responsesLite = shouldUseCodexResponsesLite(transformedBody, options?.responsesLite); + const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite); const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite); const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined; if (sessionKey && publicSessionKey) { @@ -1092,6 +1094,14 @@ async function openCodexWebSocketTransport( requestContext.responsesLite, ); const requestBodyForState = structuredCloneJSON(requestContext.transformedBody); + // `onPayload` may rewrite the outgoing frame (e.g. drop `stream_options`); + // recorded state must reflect what was actually sent — the sequential-cutoff + // summary decoder keys off it. + if (websocketRequest.stream_options === undefined) { + delete requestBodyForState.stream_options; + } else { + requestBodyForState.stream_options = websocketRequest.stream_options; + } requestContext.rawRequestDump.body = websocketRequest; CODEX_DEBUG && logger.debug("[codex] codex websocket request", { @@ -1324,6 +1334,17 @@ class CodexStreamProcessor { this.startTime = init.startTime; } + /** + * Whether the request actually sent (post-`onPayload`) opted into + * sequential-cutoff summary delivery: summaries then arrive as atomic + * `response.reasoning_summary_text.done` events and incremental + * `.delta`/`.part.*` events are ignored (mirrors codex-rs + * `uses_sequential_cutoff_reasoning_summaries`). + */ + get #sequentialCutoffSummaries(): boolean { + return this.runtime.requestBodyForState.stream_options?.reasoning_summary_delivery === "sequential_cutoff"; + } + async process(): Promise { const { output, stream } = this; stream.push({ type: "start", partial: output }); @@ -1385,6 +1406,7 @@ class CodexStreamProcessor { } if (eventType === "response.reasoning_summary_part.added") { + if (this.#sequentialCutoffSummaries) return firstTokenTime; if (this.runtime.currentItem?.type === "reasoning") { appendReasoningSummaryPart( this.runtime.currentItem, @@ -1395,6 +1417,7 @@ class CodexStreamProcessor { } if (eventType === "response.reasoning_summary_text.delta") { + if (this.#sequentialCutoffSummaries) return firstTokenTime; if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") { appendReasoningSummaryTextDelta( this.runtime.currentItem, @@ -1408,6 +1431,29 @@ class CodexStreamProcessor { return firstTokenTime; } + if (eventType === "response.reasoning_summary_text.done") { + // Outside the cutoff contract the text already streamed via `.delta`. + if (!this.#sequentialCutoffSummaries) return firstTokenTime; + const entry = this.runtime.openItemForEvent(rawEvent); + if (entry?.item.type === "reasoning" && entry.block?.type === "thinking") { + if (!firstTokenTime) firstTokenTime = performance.now(); + const summaryIndex = + typeof rawEvent.summary_index === "number" && Number.isFinite(rawEvent.summary_index) + ? Math.trunc(rawEvent.summary_index) + : 0; + applyReasoningSummaryDone( + entry.item, + entry.block, + typeof rawEvent.text === "string" ? rawEvent.text : "", + summaryIndex, + stream, + output, + entry.contentIndex, + ); + } + return firstTokenTime; + } + if (eventType === "response.reasoning_text.delta") { const entry = this.runtime.openItemForEvent(rawEvent); const delta = typeof rawEvent.delta === "string" ? rawEvent.delta : ""; @@ -1424,6 +1470,7 @@ class CodexStreamProcessor { } if (eventType === "response.reasoning_summary_part.done") { + if (this.#sequentialCutoffSummaries) return firstTokenTime; if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") { appendReasoningSummaryPartDone( this.runtime.currentItem, @@ -2142,7 +2189,7 @@ export async function prewarmOpenAICodexResponses( const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const promptCacheKey = transportSessionId; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); - const responsesLite = options?.responsesLite === true; + const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite); const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite); const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined; if (publicSessionKey && sessionKey) { @@ -3372,9 +3419,9 @@ function createCodexHeaders( headers.delete(X_MODELS_ETAG_HEADER); } if (responsesLite) { - headers.set(X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER, "true"); + headers.set(OPENAI_HEADERS.RESPONSES_LITE, "true"); } else { - headers.delete(X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER); + headers.delete(OPENAI_HEADERS.RESPONSES_LITE); } if (transport === "sse") { headers.set("accept", "text/event-stream"); diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index 968226016..8c095137f 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -35,7 +35,12 @@ export interface CodexRequestOptions { reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; - /** Responses Lite transport contract: strips image detail and disables parallel tool calling, mirroring codex-rs. */ + /** + * Responses Lite transport override; defaults to the model's + * `useResponsesLite`. Lite moves instructions/tools into input items, + * strips image detail, and disables parallel tool calling (codex-rs + * `use_responses_lite`). + */ responsesLite?: boolean; } @@ -48,6 +53,8 @@ export interface InputItem { name?: string; output?: unknown; arguments?: unknown; + /** `additional_tools` developer item payload (Responses Lite). */ + tools?: unknown; } export interface RequestBody { @@ -58,6 +65,8 @@ export interface RequestBody { input?: InputItem[]; tools?: unknown; tool_choice?: unknown; + /** Concurrent reasoning-summary delivery (codex-rs `StreamOptions`). */ + stream_options?: { reasoning_summary_delivery: "sequential_cutoff" }; // Sampling controls (temperature/top_p/top_k/min_p/presence_penalty/ // repetition_penalty/frequency_penalty/stop) are intentionally absent: the // Codex backend rejects every one with a 400 `Unsupported parameter`, so @@ -76,24 +85,16 @@ export interface RequestBody { [key: string]: unknown; } -function containsInputImage(value: unknown): boolean { - if (!value || typeof value !== "object") return false; - if ((value as { type?: unknown }).type === "input_image") return true; - if (Array.isArray(value)) { - for (const item of value) { - if (containsInputImage(item)) return true; - } - return false; - } - for (const item of Object.values(value)) { - if (containsInputImage(item)) return true; - } - return false; -} - -/** Returns whether a Codex request can use the text-only Responses Lite transport. */ -export function shouldUseCodexResponsesLite(body: RequestBody, requested: boolean | undefined): boolean { - return requested === true && !containsInputImage(body.input); +/** + * Resolve whether a Codex request uses the Responses Lite transport: an + * explicit option wins, otherwise the model's catalog flag (codex-rs + * `model_info.use_responses_lite`) decides. + */ +export function resolveCodexResponsesLite( + model: Model<"openai-codex-responses">, + requested: boolean | undefined, +): boolean { + return requested ?? model.useResponsesLite === true; } /** @@ -242,24 +243,62 @@ function repairToolCallPairs(input: InputItem[]): InputItem[] { * `detail` from every input image (message content and tool outputs) before * sending, letting the server choose. */ -function stripImageDetails(input: InputItem[]): void { +function stripImageDetails(input: unknown[]): void { for (const item of input) { - for (const collection of [item.content, item.output]) { + if (!item || typeof item !== "object") continue; + const content = "content" in item ? item.content : undefined; + const output = "output" in item ? item.output : undefined; + for (const collection of [content, output]) { if (!Array.isArray(collection)) continue; for (const part of collection) { - if ( - part && - typeof part === "object" && - (part as { type?: unknown }).type === "input_image" && - "detail" in part - ) { - part.detail = undefined; - } + if (!part || typeof part !== "object") continue; + if (!("type" in part) || part.type !== "input_image") continue; + if ("detail" in part) part.detail = undefined; } } } } +/** + * Structural view of a Responses-style body mutated by the Lite rewrite. + * Loose (`unknown`) property types let the turn transformer (`RequestBody`) + * and the agent's remote-compaction payloads reuse one shaper. + */ +export interface CodexLiteShapedBody { + instructions?: unknown; + tools?: unknown; + input?: unknown; + parallel_tool_calls?: unknown; +} + +/** + * Applies the Responses Lite body contract in place (codex-rs + * `build_responses_request` with `use_responses_lite`): strips pinned image + * detail, forces parallel tool calling off, moves tools into a leading + * `additional_tools` developer item and the base instructions into a + * developer message, then omits top-level `instructions`/`tools`. Shared by + * normal turns and both remote-compaction paths — codex-rs routes + * `/responses/compact` through the same builder. + */ +export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void { + const input = Array.isArray(body.input) ? body.input : []; + stripImageDetails(input); + body.parallel_tool_calls = false; + const prefix: InputItem[] = [ + { type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] }, + ]; + if (typeof body.instructions === "string" && body.instructions.length > 0) { + prefix.push({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: body.instructions }], + }); + } + body.input = [...prefix, ...input]; + delete body.instructions; + delete body.tools; +} + export async function transformRequestBody( body: RequestBody, model: Model<"openai-codex-responses">, @@ -333,16 +372,9 @@ export async function transformRequestBody( } } - const responsesLite = shouldUseCodexResponsesLite(body, options.responsesLite); + const responsesLite = resolveCodexResponsesLite(model, options.responsesLite); if (responsesLite) { - if (Array.isArray(body.input)) { - stripImageDetails(body.input); - } - // Responses Lite does not support parallel tool calling; codex-rs forces - // it off (`prompt.parallel_tool_calls && !use_responses_lite`). - if (body.tools !== undefined) { - body.parallel_tool_calls = false; - } + applyCodexResponsesLiteShape(body); } if (options.reasoningEffort !== undefined) { @@ -376,6 +408,17 @@ export async function transformRequestBody( body.reasoning = { ...body.reasoning, mode: model.reasoningMode }; } + // Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries` + // feature): `sequential_cutoff` lets the server stream output without + // blocking on summary generation. Only meaningful when a summary is + // requested; codex-rs additionally gates on its OpenAI provider check, + // which is inherent here. + if (body.reasoning?.summary !== undefined) { + body.stream_options = { reasoning_summary_delivery: "sequential_cutoff" }; + } else { + delete body.stream_options; + } + body.text = { ...body.text, verbosity: options.textVerbosity || "high", diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index c87069ad5..1820c24f5 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1076,6 +1076,7 @@ export const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES: ReadonlySet = new Se "response.output_item.added", "response.reasoning_summary_part.added", "response.reasoning_summary_text.delta", + "response.reasoning_summary_text.done", "response.reasoning_summary_part.done", "response.reasoning_text.delta", "response.content_part.added", @@ -1740,6 +1741,33 @@ export function appendReasoningSummaryPartDone( stream.push({ type: "thinking_delta", contentIndex, delta: "\n\n", partial: output }); } +/** + * Applies an atomic `response.reasoning_summary_text.done` event (concurrent + * reasoning summaries, `stream_options.reasoning_summary_delivery: + * "sequential_cutoff"`). The event carries the FULL text for `summaryIndex`; + * incremental `.delta`/`.part.*` events are ignored under this contract, so + * the whole part is stored and streamed here. Parts after the first are + * separated by a section break, mirroring codex-rs. + */ +export function applyReasoningSummaryDone( + item: ResponseReasoningItem, + block: ThinkingContent, + text: string, + summaryIndex: number, + stream: AssistantMessageEventStream, + output: AssistantMessage, + contentIndex: number, +): void { + item.summary = item.summary || []; + while (item.summary.length <= summaryIndex) { + item.summary.push({ type: "summary_text", text: "" }); + } + item.summary[summaryIndex].text = text; + const delta = summaryIndex > 0 ? `\n\n${text}` : text; + block.thinking += delta; + stream.push({ type: "thinking_delta", contentIndex, delta, partial: output }); +} + export function appendMessageContentPart( item: ResponseOutputMessage, part: ResponseContentPartAddedEvent["part"] | undefined, diff --git a/packages/ai/test/helpers/index.ts b/packages/ai/test/helpers/index.ts index fe987fcf1..f8437e4f8 100644 --- a/packages/ai/test/helpers/index.ts +++ b/packages/ai/test/helpers/index.ts @@ -2,6 +2,7 @@ import * as os from "node:os"; import * as path from "node:path"; import type { Model } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; import { isEnoent } from "@oh-my-pi/pi-utils"; export async function withEnv( @@ -54,7 +55,10 @@ export async function waitForDelayOrAbort(delayMs: number, signal: AbortSignal | } } -export function createCodexModel(id: string): Model<"openai-codex-responses"> { +export function createCodexModel( + id: string, + spec?: Partial>, +): Model<"openai-codex-responses"> { return buildModel({ id, name: id, @@ -66,6 +70,7 @@ export function createCodexModel(id: string): Model<"openai-codex-responses"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 272000, maxTokens: 128000, + ...spec, }); } diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index c0bd32074..f70e57d43 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -9,6 +9,7 @@ import { convertCodexResponsesMessages, streamOpenAICodexResponses, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import { isOpenAIResponsesProgressEvent } from "@oh-my-pi/pi-ai/providers/openai-shared"; import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { createCodexModel } from "./helpers"; @@ -191,7 +192,7 @@ describe("openai-codex reasoning.summary", () => { }); describe("openai-codex Responses Lite input shaping", () => { - it("keeps full Responses image details when a requested lite body contains images", async () => { + it("strips image detail and keeps lite when the input contains images", async () => { const model = createCodexModel("gpt-5.1-codex"); const makeInput = (): InputItem[] => [ { @@ -211,10 +212,11 @@ describe("openai-codex Responses Lite input shaping", () => { ]; const lite = await transformRequestBody({ model: model.id, input: makeInput() }, model, { responsesLite: true }); - const liteMessage = lite.input?.[0]?.content as Array>; - const liteOutput = lite.input?.[2]?.output as Array>; - expect(liteMessage[1]).toEqual({ type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" }); - expect(liteOutput[0]).toEqual({ type: "input_image", detail: "high", image_url: "data:image/png;base64,BBBB" }); + expect(lite.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); + const liteMessage = lite.input?.[1]?.content as Array>; + const liteOutput = lite.input?.[3]?.output as Array>; + expect(liteMessage[1]).toEqual({ type: "input_image", image_url: "data:image/png;base64,AAAA" }); + expect(liteOutput[0]).toEqual({ type: "input_image", image_url: "data:image/png;base64,BBBB" }); const plain = await transformRequestBody({ model: model.id, input: makeInput() }, model, {}); const plainMessage = plain.input?.[0]?.content as Array>; @@ -253,7 +255,7 @@ describe("openai-codex Responses Lite input shaping", () => { }); }); - it("forces parallel_tool_calls off under lite when tools are present", async () => { + it("forces parallel_tool_calls off and moves tools into input under lite", async () => { const model = createCodexModel("gpt-5.1-codex"); const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }]; @@ -261,12 +263,57 @@ describe("openai-codex Responses Lite input shaping", () => { responsesLite: true, }); expect(lite.parallel_tool_calls).toBe(false); + expect(lite.tools).toBeUndefined(); + expect(lite.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools }); const plain = await transformRequestBody({ model: model.id, tools, parallel_tool_calls: true }, model, {}); expect(plain.parallel_tool_calls).toBe(true); + expect(plain.tools).toEqual(tools); const noTools = await transformRequestBody({ model: model.id }, model, { responsesLite: true }); - expect(noTools.parallel_tool_calls).toBeUndefined(); + expect(noTools.parallel_tool_calls).toBe(false); + }); + + it("moves instructions and tools into input items under lite", async () => { + const model = createCodexModel("gpt-5.6-terra"); + const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }]; + const body = await transformRequestBody( + { + model: model.id, + instructions: "test instructions", + tools, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }], + }, + model, + { responsesLite: true }, + ); + + expect(body.instructions).toBeUndefined(); + expect(body.tools).toBeUndefined(); + expect(body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools }); + expect(body.input?.[1]).toEqual({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: "test instructions" }], + }); + expect(body.input?.[2]).toEqual({ + type: "message", + role: "user", + content: [{ type: "input_text", text: "hello" }], + }); + }); + + it("defaults lite from the model useResponsesLite flag and honors explicit opt-out", async () => { + const model = createCodexModel("gpt-5.6-terra", { useResponsesLite: true }); + const lite = await transformRequestBody({ model: model.id, instructions: "sys" }, model, {}); + expect(lite.instructions).toBeUndefined(); + expect(lite.input?.[0]?.type).toBe("additional_tools"); + + const optOut = await transformRequestBody({ model: model.id, instructions: "sys" }, model, { + responsesLite: false, + }); + expect(optOut.instructions).toBe("sys"); + expect(optOut.input?.some(item => item.type === "additional_tools")).toBe(false); }); }); @@ -342,7 +389,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); expect(captured?.body.client_metadata).toEqual(clientMetadata); }); - it("falls back to full Responses when a lite request contains images", async () => { + it("keeps lite and strips image detail when a lite request contains images", async () => { const model = buildModel({ id: "gpt-5.5", name: "GPT-5.5", @@ -382,18 +429,38 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { ).result(); expect(result.stopReason).toBe("stop"); - expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBeNull(); + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); expect(captured?.body.input).toEqual([ + { type: "additional_tools", role: "developer", tools: [] }, { role: "user", content: [ { type: "input_text", text: "read this image" }, - { type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" }, + { type: "input_image", image_url: "data:image/png;base64,AAAA" }, ], }, ]); }); + it("sends the lite header when the model defaults to Responses Lite", async () => { + const model = createCodexModel("gpt-5.6-terra", { useResponsesLite: true }); + let captured: CapturedCodexRequest | undefined; + const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { + captured = request; + }); + + const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.instructions).toBeUndefined(); + expect(captured?.body.tools).toBeUndefined(); + expect((captured?.body.input as Array>)[0]?.type).toBe("additional_tools"); + }); + it("omits the lite header and client_metadata when not requested", async () => { const model = createCodexModel("gpt-5.1-codex"); let captured: CapturedCodexRequest | undefined; @@ -467,3 +534,139 @@ describe("openai-codex websocket append with client metadata", () => { expect(transformed.client_metadata).toEqual({ "x-codex-turn-metadata": "{}" }); }); }); + +describe("openai-codex concurrent reasoning summaries", () => { + it("counts atomic summary dones as websocket watchdog progress", () => { + expect(isOpenAIResponsesProgressEvent({ type: "response.reasoning_summary_text.done" })).toBe(true); + }); + + it("sends stream_options only when a summary is requested and supported", async () => { + const terra = createCodexModel("gpt-5.6-terra"); + const withSummary = await transformRequestBody({ model: terra.id }, terra, { reasoningEffort: "medium" }); + expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); + + const suppressed = await transformRequestBody({ model: terra.id }, terra, { + reasoningEffort: "medium", + reasoningSummary: null, + }); + expect(suppressed.stream_options).toBeUndefined(); + + const noReasoning = await transformRequestBody({ model: terra.id }, terra, {}); + expect(noReasoning.stream_options).toBeUndefined(); + + const legacy = createCodexModel("gpt-5.1-codex"); + const unsupported = await transformRequestBody({ model: legacy.id }, legacy, { reasoningEffort: "medium" }); + expect(unsupported.stream_options).toBeUndefined(); + }); + + it("decodes atomic summary dones and ignores legacy deltas under sequential cutoff", async () => { + const model = createCodexModel("gpt-5.6-terra"); + const events: Array> = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "reason_1", summary: [] }, + }, + { + type: "response.reasoning_summary_part.added", + item_id: "reason_1", + output_index: 0, + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_text.delta", + item_id: "reason_1", + output_index: 0, + summary_index: 0, + delta: "IGNORED", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 0, + text: "First part", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 1, + text: "Second part", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "reasoning", + id: "reason_1", + summary: [ + { type: "summary_text", text: "First part" }, + { type: "summary_text", text: "Second part" }, + ], + }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", item_id: "msg_1", output_index: 1, delta: "Hello" }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "STALE", + }, + { + type: "response.output_item.done", + output_index: 1, + item: { + type: "message", + id: "msg_1", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "Hello" }], + }, + }, + { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + let captured: CapturedCodexRequest | undefined; + const fetchMock = createCodexFetchMock(createCodexSse(events), request => { + captured = request; + }); + + const stream = streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + reasoning: "medium", + }); + const thinkingDeltas: string[] = []; + for await (const event of stream) { + if (event.type === "thinking_delta") thinkingDeltas.push(event.delta); + } + const result = await stream.result(); + + expect(captured?.body.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); + expect(thinkingDeltas).toEqual(["First part", "\n\nSecond part"]); + expect(result.stopReason).toBe("stop"); + const thinking = result.content.find(block => block.type === "thinking"); + expect(thinking?.thinking).toBe("First part\n\nSecond part"); + const text = result.content.find(block => block.type === "text"); + expect(text?.text).toBe("Hello"); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index b733a706c..8dfc0bb26 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,19 @@ ## [Unreleased] +### Added + +- Added Grok 4.5 model family +- Added support for Dolphin Mistral 24b Venice Edition +- Added GLM5.2-Fast model +- Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) + +- Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. + +### Changed + +- Updated costs and context windows for various models in the catalog + ## [16.3.15] - 2026-07-09 ### Added diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index e8a03e0ea..5784d7088 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -30,6 +30,7 @@ const codexModelEntrySchema = type({ "supported_in_api?": "unknown", "priority?": "unknown", "prefer_websockets?": "unknown", + "use_responses_lite?": "unknown", }); const codexModelsResponseSchema = type({ @@ -262,6 +263,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels); const input = normalizeInputModalities(payload.input_modalities); const preferWebsockets = toBoolean(payload.prefer_websockets) === true; + const useResponsesLite = toBoolean(payload.use_responses_lite) === true; const priority = toFiniteNumber(payload.priority) ?? Number.MAX_SAFE_INTEGER; return { @@ -279,6 +281,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo contextWindow, maxTokens, ...(preferWebsockets ? { preferWebsockets: true } : {}), + ...(useResponsesLite ? { useResponsesLite: true } : {}), ...(priority !== Number.MAX_SAFE_INTEGER ? { priority } : {}), }, }; diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 0d7cc7d3d..706e3b509 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -17454,6 +17454,69 @@ }, "requestModelId": "gpt-5-6-terra-none-priority" }, + "grok-4-5-high": { + "id": "grok-4-5-high", + "name": "Grok 4.5 High", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 64000 + }, + "grok-4-5-low": { + "id": "grok-4-5-low", + "name": "Grok 4.5 Low", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 64000 + }, + "grok-4-5-medium": { + "id": "grok-4-5-medium", + "name": "Grok 4.5 Medium", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 64000 + }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", @@ -26266,6 +26329,25 @@ "contextWindow": null, "maxTokens": null }, + "cognitivecomputations/dolphin-mistral-24b-venice-edition": { + "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition", + "name": "Uncensored", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "cohere/command-a": { "id": "cohere/command-a", "name": "Command A", @@ -61361,7 +61443,7 @@ "contextWindow": 272000, "maxTokens": 128000, "preferWebsockets": true, - "priority": 0, + "priority": 7, "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -61399,6 +61481,7 @@ "contextWindow": 372000, "maxTokens": 128000, "preferWebsockets": true, + "useResponsesLite": true, "priority": 3, "applyPatchToolType": "freeform", "thinking": { @@ -61444,6 +61527,7 @@ "contextWindow": 372000, "maxTokens": 128000, "preferWebsockets": true, + "useResponsesLite": true, "priority": 3, "requestModelId": "gpt-5.6-luna", "reasoningMode": "pro", @@ -61491,6 +61575,7 @@ "contextWindow": 372000, "maxTokens": 128000, "preferWebsockets": true, + "useResponsesLite": true, "priority": 1, "applyPatchToolType": "freeform", "thinking": { @@ -61536,6 +61621,7 @@ "contextWindow": 372000, "maxTokens": 128000, "preferWebsockets": true, + "useResponsesLite": true, "priority": 1, "requestModelId": "gpt-5.6-sol", "reasoningMode": "pro", @@ -61583,6 +61669,7 @@ "contextWindow": 372000, "maxTokens": 128000, "preferWebsockets": true, + "useResponsesLite": true, "priority": 2, "applyPatchToolType": "freeform", "thinking": { @@ -61628,6 +61715,7 @@ "contextWindow": 372000, "maxTokens": 128000, "preferWebsockets": true, + "useResponsesLite": true, "priority": 2, "requestModelId": "gpt-5.6-terra", "reasoningMode": "pro", @@ -64649,7 +64737,7 @@ "image" ], "cost": { - "input": 0.65, + "input": 0.66, "output": 3.41, "cacheRead": 0.15, "cacheWrite": 0 @@ -66320,9 +66408,9 @@ "text" ], "cost": { - "input": 0.2288, - "output": 0.3432, - "cacheRead": 0.02288, + "input": 0.2145, + "output": 0.32175, + "cacheRead": 0.02145, "cacheWrite": 0 }, "contextWindow": 131072, @@ -68450,7 +68538,7 @@ "image" ], "cost": { - "input": 0.65, + "input": 0.66, "output": 3.41, "cacheRead": 0.15, "cacheWrite": 0 @@ -73884,13 +73972,13 @@ "text" ], "cost": { - "input": 0.54, - "output": 1.76, - "cacheRead": 0.09999999999999999, + "input": 0.9199999999999999, + "output": 3, + "cacheRead": 0.18, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 101376, + "maxTokens": 1048576, "thinking": { "mode": "effort", "efforts": [ @@ -77771,134 +77859,182 @@ }, "openai-gpt-56-luna": { "id": "openai-gpt-56-luna", - "name": "openai-gpt-56-luna", + "name": "GPT-5.6 Luna", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.25, + "output": 7.5, + "cacheRead": 0.125, "cacheWrite": 0 }, "contextWindow": 1000000, "maxTokens": 128000, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "openai-gpt-56-luna-pro": { "id": "openai-gpt-56-luna-pro", - "name": "openai-gpt-56-luna-pro", + "name": "GPT-5.6 Luna Pro", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.25, + "output": 7.5, + "cacheRead": 0.125, "cacheWrite": 0 }, "contextWindow": 1000000, "maxTokens": 128000, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "openai-gpt-56-sol": { "id": "openai-gpt-56-sol", - "name": "openai-gpt-56-sol", + "name": "GPT-5.6 Sol", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 6.25, + "output": 37.5, + "cacheRead": 0.625, "cacheWrite": 0 }, "contextWindow": 1000000, "maxTokens": 128000, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "openai-gpt-56-sol-pro": { "id": "openai-gpt-56-sol-pro", - "name": "openai-gpt-56-sol-pro", + "name": "GPT-5.6 Sol Pro", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 6.25, + "output": 37.5, + "cacheRead": 0.625, "cacheWrite": 0 }, "contextWindow": 1000000, "maxTokens": 128000, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "openai-gpt-56-terra": { "id": "openai-gpt-56-terra", - "name": "openai-gpt-56-terra", + "name": "GPT-5.6 Terra", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 3.125, + "output": 18.75, + "cacheRead": 0.3125, "cacheWrite": 0 }, "contextWindow": 1000000, "maxTokens": 128000, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "openai-gpt-56-terra-pro": { "id": "openai-gpt-56-terra-pro", - "name": "openai-gpt-56-terra-pro", + "name": "GPT-5.6 Terra Pro", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 3.125, + "output": 18.75, + "cacheRead": 0.3125, "cacheWrite": 0 }, "contextWindow": 1000000, "maxTokens": 128000, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "openai-gpt-oss-120b": { @@ -80221,7 +80357,7 @@ "cost": { "input": 0.14, "output": 0.28, - "cacheRead": 0.0028, + "cacheRead": 0.028, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -81018,7 +81154,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 1.25, @@ -84607,6 +84744,39 @@ "supportsDeveloperRole": false } }, + "glm5.2-fast": { + "id": "glm5.2-fast", + "name": "GLM5.2-Fast", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3.75, + "output": 12.8125, + "cacheRead": 0.625, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "compat": { + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + } + }, "GLM5.2-Turbo": { "id": "GLM5.2-Turbo", "name": "GLM5.2-Turbo", @@ -85705,6 +85875,19 @@ }, "contextWindow": 500000, "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -90070,6 +90253,117 @@ }, "contextPromotionTarget": "zenmux/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "openai/gpt-image-1.5": { "id": "openai/gpt-image-1.5", "name": "GPT-Image-1.5", diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index c0f4d7620..0407a2050 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -758,6 +758,8 @@ export interface Model { transport?: "pi-native"; /** Hint that websocket transport should be preferred when supported by the provider implementation. */ preferWebsockets?: boolean; + /** Codex Responses Lite transport: send the lite marker and carry instructions/tools as input items (mirrors codex-rs `use_responses_lite`). */ + useResponsesLite?: boolean; /** Preferred model to switch to when context promotion is triggered (model id or provider/id). */ contextPromotionTarget?: string; /** Preferred model to use only for compaction (model id or provider/id); the active session model is unchanged. */ diff --git a/packages/catalog/src/wire/codex.ts b/packages/catalog/src/wire/codex.ts index 1d3d80700..c2336b6d2 100644 --- a/packages/catalog/src/wire/codex.ts +++ b/packages/catalog/src/wire/codex.ts @@ -10,6 +10,8 @@ export const OPENAI_HEADERS = { ORIGINATOR: "originator", SESSION_ID: "session_id", CONVERSATION_ID: "conversation_id", + /** Responses Lite transport marker (codex-rs `add_responses_lite_header`); value is always `"true"`. */ + RESPONSES_LITE: "x-openai-internal-codex-responses-lite", } as const; export const OPENAI_HEADER_VALUES = { diff --git a/packages/catalog/test/codex-discovery.test.ts b/packages/catalog/test/codex-discovery.test.ts index b6f87c79f..babff9c28 100644 --- a/packages/catalog/test/codex-discovery.test.ts +++ b/packages/catalog/test/codex-discovery.test.ts @@ -52,6 +52,50 @@ describe("Codex model discovery", () => { }); }); + it("carries use_responses_lite and prefer_websockets onto the model spec", async () => { + const fetchFn: typeof fetch = Object.assign( + async () => + new Response( + JSON.stringify({ + models: [ + { + slug: "gpt-5.6-terra", + display_name: "GPT-5.6-Terra", + context_window: 372_000, + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + prefer_websockets: true, + use_responses_lite: true, + }, + { + slug: "gpt-5.5", + display_name: "GPT-5.5", + context_window: 272_000, + default_reasoning_level: "high", + supported_reasoning_levels: ["low", "high"], + input_modalities: ["text"], + supported_in_api: true, + }, + ], + }), + ), + { preconnect() {} }, + ); + const result = await fetchCodexModels({ + accessToken: "test-token", + baseUrl: "https://codex.example/backend-api", + clientVersion: "0.99.0", + fetchFn, + }); + + const terra = result?.models.find(model => model.id === "gpt-5.6-terra"); + expect(terra).toMatchObject({ preferWebsockets: true, useResponsesLite: true }); + const legacy = result?.models.find(model => model.id === "gpt-5.5"); + expect(legacy?.useResponsesLite).toBeUndefined(); + }); + it("ignores pre-V2 Codex discovery cache rows", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-v7-cache-")); const dbPath = path.join(tempDir, "models.db"); From a45ecd55939fc443294b4d011095b0ef2976103d Mon Sep 17 00:00:00 2001 From: freecodewu Date: Thu, 9 Jul 2026 19:46:11 +0800 Subject: [PATCH 050/205] Add Novita provider --- packages/ai/src/registry/novita.ts | 21 + packages/ai/src/registry/registry.ts | 2 + packages/catalog/src/models.json | 2865 ++++++++++++++++- .../src/provider-models/descriptors.ts | 8 + .../src/provider-models/openai-compat.ts | 94 + packages/catalog/test/novita-provider.test.ts | 97 + 6 files changed, 3086 insertions(+), 1 deletion(-) create mode 100644 packages/ai/src/registry/novita.ts create mode 100644 packages/catalog/test/novita-provider.test.ts diff --git a/packages/ai/src/registry/novita.ts b/packages/ai/src/registry/novita.ts new file mode 100644 index 000000000..3b573dab6 --- /dev/null +++ b/packages/ai/src/registry/novita.ts @@ -0,0 +1,21 @@ +import { createApiKeyLogin } from "./api-key-login"; +import type { ProviderDefinition } from "./types"; + +export const loginNovita = createApiKeyLogin({ + providerLabel: "Novita", + authUrl: "https://novita.ai/settings/key-management", + instructions: "Create or copy your API key from the Novita dashboard", + promptMessage: "Paste your Novita API key", + placeholder: "sk_...", + validation: { + kind: "models-endpoint", + provider: "novita", + modelsUrl: "https://api.novita.ai/openai/v1/models", + }, +}); + +export const novitaProvider = { + id: "novita", + name: "Novita", + login: (cb: Parameters[0]) => loginNovita(cb), +} as const satisfies ProviderDefinition; diff --git a/packages/ai/src/registry/registry.ts b/packages/ai/src/registry/registry.ts index e6740996a..2d9420c49 100644 --- a/packages/ai/src/registry/registry.ts +++ b/packages/ai/src/registry/registry.ts @@ -34,6 +34,7 @@ import { minimaxCodeCnProvider } from "./minimax-code-cn"; import { mistralProvider } from "./mistral"; import { moonshotProvider } from "./moonshot"; import { nanogptProvider } from "./nanogpt"; +import { novitaProvider } from "./novita"; import { nvidiaProvider } from "./nvidia"; import { ollamaProvider } from "./ollama"; import { ollamaCloudProvider } from "./ollama-cloud"; @@ -110,6 +111,7 @@ const ALL = [ fireworksProvider, togetherProvider, nvidiaProvider, + novitaProvider, huggingfaceProvider, perplexityProvider, qianfanProvider, diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 706e3b509..a4e4451e8 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -58324,6 +58324,2869 @@ "maxTokens": null } }, + "novita": { + "baichuan/baichuan-m2-32b": { + "id": "baichuan/baichuan-m2-32b", + "name": "BaiChuan M2 32B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 0.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072, + "supportsTools": false + }, + "baidu/cobuddy": { + "id": "baidu/cobuddy", + "name": "CoBuddy", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.8, + "output": 11.3, + "cacheRead": 0.7, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "baidu/ernie-4.5-21B-a3b": { + "id": "baidu/ernie-4.5-21B-a3b", + "name": "ERNIE 4.5 21B A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 2.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 120000, + "maxTokens": 8000, + "supportsTools": true + }, + "baidu/ernie-4.5-vl-424b-a47b": { + "id": "baidu/ernie-4.5-vl-424b-a47b", + "name": "ERNIE 4.5 VL 424B A47B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 4.2, + "output": 12.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 123000, + "maxTokens": 16000, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "bunny": { + "id": "bunny", + "name": "Bunny", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "deepseek/deepseek_v3": { + "id": "deepseek/deepseek_v3", + "name": "DeepSeek V3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 8.9, + "output": 8.9, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true + }, + "deepseek/deepseek-ocr": { + "id": "deepseek/deepseek-ocr", + "name": "DeepSeek-OCR", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 0.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "deepseek/deepseek-ocr-2": { + "id": "deepseek/deepseek-ocr-2", + "name": "DeepSeek-OCR 2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 0.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "deepseek/deepseek-r1": { + "id": "deepseek/deepseek-r1", + "name": "R1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 40, + "output": 40, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-0528": { + "id": "deepseek/deepseek-r1-0528", + "name": "R1 0528", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 7, + "output": 25, + "cacheRead": 3.5, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-0528-qwen3-8b": { + "id": "deepseek/deepseek-r1-0528-qwen3-8b", + "name": "DeepSeek R1 0528 Qwen3 8B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 0.9, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32000, + "supportsTools": false + }, + "deepseek/deepseek-r1-distill-llama-70b": { + "id": "deepseek/deepseek-r1-distill-llama-70b", + "name": "DeepSeek R1 Distill LLama 70B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 8, + "output": 8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-turbo": { + "id": "deepseek/deepseek-r1-turbo", + "name": "DeepSeek R1 (Turbo)", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 7, + "output": 25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1/community": { + "id": "deepseek/deepseek-r1/community", + "name": "DeepSeek R1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 40, + "output": 40, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 8000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3-0324": { + "id": "deepseek/deepseek-v3-0324", + "name": "DeepSeek V3 0324", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 2.7, + "output": 11.2, + "cacheRead": 1.35, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true + }, + "deepseek/deepseek-v3-turbo": { + "id": "deepseek/deepseek-v3-turbo", + "name": "DeepSeek V3 (Turbo)", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 4, + "output": 13, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true + }, + "deepseek/deepseek-v3.1": { + "id": "deepseek/deepseek-v3.1", + "name": "DeepSeek V3.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.7, + "output": 10, + "cacheRead": 1.35, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.1-terminus": { + "id": "deepseek/deepseek-v3.1-terminus", + "name": "DeepSeek V3.1 Terminus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.7, + "output": 10, + "cacheRead": 1.35, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.2": { + "id": "deepseek/deepseek-v3.2", + "name": "DeepSeek V3.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.69, + "output": 4, + "cacheRead": 1.345, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.2-exp": { + "id": "deepseek/deepseek-v3.2-exp", + "name": "DeepSeek V3.2 Exp", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.7, + "output": 4.1, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3/community": { + "id": "deepseek/deepseek-v3/community", + "name": "DeepSeek V3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 8.9, + "output": 8.9, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 8000, + "supportsTools": true + }, + "deepseek/deepseek-v4-flash": { + "id": "deepseek/deepseek-v4-flash", + "name": "DeepSeek V4 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 2.8, + "cacheRead": 0.28, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v4-pro": { + "id": "deepseek/deepseek-v4-pro", + "name": "DeepSeek V4 Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 16, + "output": 32, + "cacheRead": 1.35, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "dev/glm46": { + "id": "dev/glm46", + "name": "dev/glm46", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 256000, + "supportsTools": true + }, + "google/gemma-3-12b-it": { + "id": "google/gemma-3-12b-it", + "name": "Gemma3 12B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.5, + "output": 1, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 8192, + "supportsTools": false + }, + "google/gemma-3-27b-it": { + "id": "google/gemma-3-27b-it", + "name": "Gemma 3 27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.19, + "output": 2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 98304, + "maxTokens": 16384, + "supportsTools": false + }, + "google/gemma-4-26b-a4b-it": { + "id": "google/gemma-4-26b-a4b-it", + "name": "Gemma 4 26B A4B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.3, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "google/gemma-4-31b-it": { + "id": "google/gemma-4-31b-it", + "name": "Gemma 4 31B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.4, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gryphe/mythomax-l2-13b": { + "id": "gryphe/mythomax-l2-13b", + "name": "Mythomax L2 13B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.9, + "output": 0.9, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 4096, + "maxTokens": 3200, + "supportsTools": false + }, + "gt-4p": { + "id": "gt-4p", + "name": "gt-4p", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": 131072, + "supportsTools": true + }, + "inclusionai/ling-2.6-1t": { + "id": "inclusionai/ling-2.6-1t", + "name": "Ling-2.6-1T", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 25, + "cacheRead": 0.6, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "inclusionai/ling-2.6-flash": { + "id": "inclusionai/ling-2.6-flash", + "name": "Ling-2.6 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1, + "output": 3, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "inclusionai/ring-2.6-1t": { + "id": "inclusionai/ring-2.6-1t", + "name": "Ring-2.6-1T", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 25, + "cacheRead": 0.6, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "kwaipilot/kat-coder-pro": { + "id": "kwaipilot/kat-coder-pro", + "name": "Kat Coder Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 12, + "cacheRead": 0.6, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 128000, + "supportsTools": true + }, + "meta-llama/llama-3.1-8b-instruct": { + "id": "meta-llama/llama-3.1-8b-instruct", + "name": "Llama 3.1 8B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 0.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 16384, + "supportsTools": false + }, + "meta-llama/llama-3.2-1b-instruct": { + "id": "meta-llama/llama-3.2-1b-instruct", + "name": "Llama 3.2 1B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 0.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131000, + "maxTokens": 32000, + "supportsTools": false + }, + "meta-llama/llama-3.2-3b-instruct": { + "id": "meta-llama/llama-3.2-3b-instruct", + "name": "Llama 3.2 3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 0.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32768, + "maxTokens": 32000, + "supportsTools": false + }, + "meta-llama/llama-3.3-70b-instruct": { + "id": "meta-llama/llama-3.3-70b-instruct", + "name": "Llama 3.3 70B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1.35, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 6000, + "maxTokens": 120000, + "supportsTools": true + }, + "meta-llama/llama-4-maverick-17b-128e-instruct-fp8": { + "id": "meta-llama/llama-4-maverick-17b-128e-instruct-fp8", + "name": "Llama 4 Maverick Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.7, + "output": 8.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 8192, + "supportsTools": false + }, + "meta-llama/llama-4-scout-17b-16e-instruct": { + "id": "meta-llama/llama-4-scout-17b-16e-instruct", + "name": "Llama 4 Scout 17B 16E", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.8, + "output": 5.9, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072, + "supportsTools": false + }, + "microsoft/wizardlm-2-8x22b": { + "id": "microsoft/wizardlm-2-8x22b", + "name": "Wizardlm 2 8x22B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 6.2, + "output": 6.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65535, + "maxTokens": 8000, + "supportsTools": false + }, + "minimax/m2-her": { + "id": "minimax/m2-her", + "name": "M2-her", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": null, + "supportsTools": false + }, + "minimax/minimax-m2": { + "id": "minimax/minimax-m2", + "name": "MiniMax M2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 12, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.1": { + "id": "minimax/minimax-m2.1", + "name": "MiniMax M2.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 12, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.5": { + "id": "minimax/minimax-m2.5", + "name": "MiniMax M2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 12, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131100, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.5-highspeed": { + "id": "minimax/minimax-m2.5-highspeed", + "name": "MiniMax M2.5-highspeed", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 6, + "output": 24, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131100, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.7": { + "id": "minimax/minimax-m2.7", + "name": "MiniMax M2.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 12, + "cacheRead": 0.6, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.7-highspeed": { + "id": "minimax/minimax-m2.7-highspeed", + "name": "MiniMax M2.7 highspeed", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 6, + "output": 24, + "cacheRead": 0.6, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax M3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 12, + "cacheRead": 0.6, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "minimaxai/minimax-m1-80k": { + "id": "minimaxai/minimax-m1-80k", + "name": "MiniMax M1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 5.5, + "output": 22, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 40000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "mistralai/mistral-nemo": { + "id": "mistralai/mistral-nemo", + "name": "Mistral Nemo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.4, + "output": 1.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 60288, + "maxTokens": 16000, + "supportsTools": false + }, + "moonshotai/kimi-k2-0905": { + "id": "moonshotai/kimi-k2-0905", + "name": "Kimi K2 0905", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 6, + "output": 25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 100352, + "supportsTools": true + }, + "moonshotai/kimi-k2-instruct": { + "id": "moonshotai/kimi-k2-instruct", + "name": "Kimi K2 Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 5.7, + "output": 23, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 100352, + "supportsTools": true + }, + "moonshotai/kimi-k2-thinking": { + "id": "moonshotai/kimi-k2-thinking", + "name": "Kimi K2 Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 6, + "output": 25, + "cacheRead": 1.5, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 100352, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "moonshotai/kimi-k2.5": { + "id": "moonshotai/kimi-k2.5", + "name": "Kimi K2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 6, + "output": 30, + "cacheRead": 1, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/kimi-k2.6": { + "id": "moonshotai/kimi-k2.6", + "name": "Kimi K2.6", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 8, + "output": 34, + "cacheRead": 1.6, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/kimi-k2.7-code": { + "id": "moonshotai/kimi-k2.7-code", + "name": "Kimi K2.7 Code", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 9.5, + "output": 40, + "cacheRead": 1.9, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "nousresearch/hermes-2-pro-llama-3-8b": { + "id": "nousresearch/hermes-2-pro-llama-3-8b", + "name": "Hermes 2 Pro Llama 3 8B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 1.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "nvidia/nemotron-3-nano-30b-a3b": { + "id": "nvidia/nemotron-3-nano-30b-a3b", + "name": "Nemotron 3 Nano 30B A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.5, + "output": 2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-oss-120b": { + "id": "openai/gpt-oss-120b", + "name": "GPT OSS 120B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.5, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "openai/gpt-oss-20b": { + "id": "openai/gpt-oss-20b", + "name": "GPT OSS 20B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.4, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "paddlepaddle/paddleocr-vl": { + "id": "paddlepaddle/paddleocr-vl", + "name": "PaddleOCR-VL", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.2, + "output": 0.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 16384, + "supportsTools": false + }, + "qwen/qwen-2.5-72b-instruct": { + "id": "qwen/qwen-2.5-72b-instruct", + "name": "Qwen2.5 72B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 3.8, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": 8192, + "supportsTools": true + }, + "qwen/qwen-mt-plus": { + "id": "qwen/qwen-mt-plus", + "name": "Qwen MT Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 2.5, + "output": 7.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 8192, + "supportsTools": false + }, + "qwen/qwen3-235b-a22b-fp8": { + "id": "qwen/qwen3-235b-a22b-fp8", + "name": "Qwen3 235B A22B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2, + "output": 8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 40960, + "maxTokens": 20000, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3-235b-a22b-instruct-2507": { + "id": "qwen/qwen3-235b-a22b-instruct-2507", + "name": "Qwen3 235B A22B Instruct 2507", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.9, + "output": 5.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 16384, + "supportsTools": true + }, + "qwen/qwen3-235b-a22b-thinking-2507": { + "id": "qwen/qwen3-235b-a22b-thinking-2507", + "name": "Qwen3 235B A22B Thinking 2507", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 30, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-coder-30b-a3b-instruct": { + "id": "qwen/qwen3-coder-30b-a3b-instruct", + "name": "Qwen3 Coder 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 2.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 160000, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-coder-480b-a35b-instruct": { + "id": "qwen/qwen3-coder-480b-a35b-instruct", + "name": "Qwen3 Coder 480B A35B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 3.8, + "output": 15.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-coder-next": { + "id": "qwen/qwen3-coder-next", + "name": "Qwen3 Coder Next", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 2, + "output": 15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-max": { + "id": "qwen/qwen3-max", + "name": "Qwen3 Max", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 21.1, + "output": 84.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-next-80b-a3b-instruct": { + "id": "qwen/qwen3-next-80b-a3b-instruct", + "name": "Qwen3-Next-80B-A3B-Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1.5, + "output": 15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-omni-30b-a3b-instruct": { + "id": "qwen/qwen3-omni-30b-a3b-instruct", + "name": "Qwen3 Omni 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 9.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true + }, + "qwen/qwen3-omni-30b-a3b-thinking": { + "id": "qwen/qwen3-omni-30b-a3b-thinking", + "name": "Qwen3 Omni 30B A3B Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 9.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-vl-235b-a22b-instruct": { + "id": "qwen/qwen3-vl-235b-a22b-instruct", + "name": "Qwen3 VL 235B A22B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-vl-235b-a22b-thinking": { + "id": "qwen/qwen3-vl-235b-a22b-thinking", + "name": "Qwen3 VL 235B A22B Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 9.8, + "output": 39.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-vl-30b-a3b-instruct": { + "id": "qwen/qwen3-vl-30b-a3b-instruct", + "name": "Qwen3 VL 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3.5-122b-a10b": { + "id": "qwen/qwen3.5-122b-a10b", + "name": "Qwen3.5 122B-A10B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 4, + "output": 32, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-27b": { + "id": "qwen/qwen3.5-27b", + "name": "Qwen3.5-27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 24, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-35b-a3b": { + "id": "qwen/qwen3.5-35b-a3b", + "name": "Qwen3.5-35B-A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 20, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-397b-a17b": { + "id": "qwen/qwen3.5-397b-a17b", + "name": "Qwen3.5 397B A17B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 6, + "output": 36, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-plus": { + "id": "qwen/qwen3.5-plus", + "name": "Qwen3.5 Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-27b": { + "id": "qwen/qwen3.6-27b", + "name": "Qwen3.6 27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 6, + "output": 36, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-35b-a3b": { + "id": "qwen/qwen3.6-35b-a3b", + "name": "Qwen3.6-35B-A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.48, + "output": 14.85, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-plus": { + "id": "qwen/qwen3.6-plus", + "name": "Qwen3.6 Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.7-max": { + "id": "qwen/qwen3.7-max", + "name": "Qwen3.7 Max", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 12.5, + "output": 37.5, + "cacheRead": 2.5, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "sao10k/l3-70b-euryale-v2.1": { + "id": "sao10k/l3-70b-euryale-v2.1", + "name": "L3 70B Euryale V2.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 14.8, + "output": 14.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": true + }, + "sao10k/l3-8b-lunaris": { + "id": "sao10k/l3-8b-lunaris", + "name": "Sao10k L3 8B Lunaris", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.5, + "output": 0.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "Sao10K/L3-8B-Stheno-v3.2": { + "id": "Sao10K/L3-8B-Stheno-v3.2", + "name": "L3 8B Stheno V3.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.5, + "output": 0.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 32000, + "supportsTools": true + }, + "sao10k/l31-70b-euryale-v2.2": { + "id": "sao10k/l31-70b-euryale-v2.2", + "name": "L31 70B Euryale V2.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 14.8, + "output": 14.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "stepfun/step-3.7-flash": { + "id": "stepfun/step-3.7-flash", + "name": "Step 3.7 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 11.5, + "cacheRead": 0.4, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 256000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "thudm/glm-4-32b-0414": { + "id": "thudm/glm-4-32b-0414", + "name": "GLM-4-32B-0414", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 5.5, + "output": 16.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": 32000, + "supportsTools": true + }, + "xiaomimimo/mimo-v2.5": { + "id": "xiaomimimo/mimo-v2.5", + "name": "XiaomiMiMo/MiMo-V2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.68, + "output": 3.36, + "cacheRead": 0.034, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "xiaomimimo/mimo-v2.5-pro": { + "id": "xiaomimimo/mimo-v2.5-pro", + "name": "XiaomiMiMo/MiMo-V2.5-Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 5.22, + "output": 10.44, + "cacheRead": 0.043, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "zai-org/autoglm-phone-9b-multilingual": { + "id": "zai-org/autoglm-phone-9b-multilingual", + "name": "AutoGLM-Phone-9B-Multilingual", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.35, + "output": 1.38, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 65536, + "supportsTools": false + }, + "zai-org/glm-4.5-air": { + "id": "zai-org/glm-4.5-air", + "name": "zai-org/glm-4.5-air", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.3, + "output": 8.5, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 98304, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.5v": { + "id": "zai-org/glm-4.5v", + "name": "GLM 4.5V", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 6, + "output": 18, + "cacheRead": 1.1, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.6": { + "id": "zai-org/glm-4.6", + "name": "GLM 4.6", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 5.5, + "output": 22, + "cacheRead": 1.1, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.6v": { + "id": "zai-org/glm-4.6v", + "name": "GLM 4.6V", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 9, + "cacheRead": 0.55, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7": { + "id": "zai-org/glm-4.7", + "name": "GLM 4.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 6, + "output": 22, + "cacheRead": 1.1, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7-flash": { + "id": "zai-org/glm-4.7-flash", + "name": "GLM 4.7 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 4, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 128000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7-h": { + "id": "zai-org/glm-4.7-h", + "name": "GLM-4.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 6, + "output": 22, + "cacheRead": 1.1, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5": { + "id": "zai-org/glm-5", + "name": "GLM 5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 10, + "output": 32, + "cacheRead": 2, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5-turbo": { + "id": "zai-org/glm-5-turbo", + "name": "GLM-5-Turbo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 12, + "output": 40, + "cacheRead": 2.4, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5.1": { + "id": "zai-org/glm-5.1", + "name": "GLM 5.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 13.8, + "output": 44, + "cacheRead": 2.6, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5.2": { + "id": "zai-org/glm-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 14, + "output": 44, + "cacheRead": 2.6, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "xhigh": "max" + } + } + }, + "zai-org/glm-5v-turbo": { + "id": "zai-org/glm-5v-turbo", + "name": "GLM-5V-Turbo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 12, + "output": 40, + "cacheRead": 2.4, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + } + }, "ollama-cloud": { "cogito-2.1:671b": { "id": "cogito-2.1:671b", @@ -92710,4 +95573,4 @@ } } } -} \ No newline at end of file +} diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index 29a80ecf4..f993a6538 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -29,6 +29,7 @@ import { mistralModelManagerOptions, moonshotModelManagerOptions, nanoGptModelManagerOptions, + novitaModelManagerOptions, nvidiaModelManagerOptions, ollamaModelManagerOptions, openaiModelManagerOptions, @@ -271,6 +272,13 @@ export const CATALOG_PROVIDERS = [ createModelManagerOptions: (config: ModelManagerConfig) => nvidiaModelManagerOptions(config), catalogDiscovery: { label: "NVIDIA" }, }, + { + id: "novita", + defaultModel: "moonshotai/kimi-k2.7-code", + envVars: ["NOVITA_API_KEY"], + createModelManagerOptions: (config: ModelManagerConfig) => novitaModelManagerOptions(config), + catalogDiscovery: { label: "Novita", allowUnauthenticated: true }, + }, { id: "ollama", defaultModel: "gpt-oss:20b", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index b3e4af403..2e8cd4438 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -971,6 +971,100 @@ export function nvidiaModelManagerOptions( return createSimpleOpenAICompletionsOptions("nvidia", "https://integrate.api.nvidia.com/v1", config); } +// --------------------------------------------------------------------------- +// 5.5 Novita +// --------------------------------------------------------------------------- + +export interface NovitaModelManagerConfig { + apiKey?: string; + baseUrl?: string; + fetch?: FetchImpl; +} + +function novitaArrayIncludes(value: unknown, expected: string): boolean { + return Array.isArray(value) && value.some(item => item === expected); +} + +function isPublicNovitaModelId(id: string): boolean { + return !id.toLowerCase().startsWith("ai_infer_test"); +} + +function toNovitaCostPerMillion(value: unknown): number { + return toPositiveNumber(value, 0) / 1000; +} + +function getNovitaCacheReadPricePerMillion(entry: OpenAICompatibleModelRecord): number { + const pricing = entry.pricing; + if (!isRecord(pricing)) { + return 0; + } + const cacheRead = pricing.input_cache_read; + if (!isRecord(cacheRead)) { + return 0; + } + return toNovitaCostPerMillion(cacheRead.price_per_m); +} + +function mapNovitaModel( + entry: OpenAICompatibleModelRecord, + defaults: ModelSpec<"openai-completions">, + reference: ModelSpec<"openai-completions"> | undefined, +): ModelSpec<"openai-completions"> { + const model = mapWithBundledReference( + { + ...entry, + name: entry.display_name ?? entry.title ?? entry.name, + }, + defaults, + reference, + ); + return { + ...model, + reasoning: novitaArrayIncludes(entry.features, "reasoning"), + supportsTools: novitaArrayIncludes(entry.features, "function-calling"), + input: toInputCapabilities(entry.input_modalities), + cost: { + input: toNovitaCostPerMillion(entry.input_token_price_per_m), + output: toNovitaCostPerMillion(entry.output_token_price_per_m), + cacheRead: getNovitaCacheReadPricePerMillion(entry), + cacheWrite: 0, + }, + contextWindow: toPositiveNumber(entry.context_size, model.contextWindow), + maxTokens: toPositiveNumber(entry.max_output_tokens, model.maxTokens), + }; +} + +export function novitaModelManagerOptions( + config?: NovitaModelManagerConfig, +): ModelManagerOptions<"openai-completions"> { + const apiKey = config?.apiKey; + const baseUrl = config?.baseUrl ?? "https://api.novita.ai/openai/v1"; + const references = createBundledReferenceMap<"openai-completions">( + "novita" as Parameters[0], + ); + return { + providerId: "novita", + fetchDynamicModels: async () => + fetchOpenAICompatibleModels({ + api: "openai-completions", + provider: "novita", + baseUrl, + apiKey, + mapModel: (entry, defaults) => mapNovitaModel(entry, defaults, references.get(defaults.id)), + filterModel: (entry, model) => { + const active = typeof entry.status !== "number" || entry.status === 1; + return ( + active && + isPublicNovitaModelId(model.id) && + novitaArrayIncludes(entry.endpoints, "chat/completions") && + model.maxTokens !== 0 + ); + }, + fetch: config?.fetch, + }), + }; +} + // --------------------------------------------------------------------------- // 6. xAI // --------------------------------------------------------------------------- diff --git a/packages/catalog/test/novita-provider.test.ts b/packages/catalog/test/novita-provider.test.ts new file mode 100644 index 000000000..f737ec9dc --- /dev/null +++ b/packages/catalog/test/novita-provider.test.ts @@ -0,0 +1,97 @@ +import { describe, expect, test } from "bun:test"; +import { getOAuthProviders } from "@oh-my-pi/pi-ai/registry/oauth"; +import { getEnvApiKey } from "@oh-my-pi/pi-ai/stream"; +import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "@oh-my-pi/pi-catalog/provider-models/descriptors"; +import { novitaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; + +describe("Novita built-in provider", () => { + test("registers catalog descriptor with NOVITA_API_KEY env discovery", () => { + const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "novita"); + expect(descriptor).toBeDefined(); + expect(descriptor?.defaultModel).toBe("moonshotai/kimi-k2.7-code"); + expect(descriptor?.catalogDiscovery?.envVars).toContain("NOVITA_API_KEY"); + expect(descriptor?.catalogDiscovery?.allowUnauthenticated).toBe(true); + expect(DEFAULT_MODEL_PER_PROVIDER.novita).toBe("moonshotai/kimi-k2.7-code"); + }); + + test("registers Novita as an API-key login provider", () => { + const provider = getOAuthProviders().find(item => item.id === "novita"); + expect(provider?.name).toBe("Novita"); + expect(provider?.available).toBe(true); + }); + + test("resolves NOVITA_API_KEY via env", () => { + const previous = Bun.env.NOVITA_API_KEY; + Bun.env.NOVITA_API_KEY = "novita-test-key"; + try { + expect(getEnvApiKey("novita")).toBe("novita-test-key"); + } finally { + if (previous === undefined) { + delete Bun.env.NOVITA_API_KEY; + } else { + Bun.env.NOVITA_API_KEY = previous; + } + } + }); + + test("maps Novita model catalog metadata from the public OpenAI-compatible endpoint", async () => { + const requests: string[] = []; + const fetchMock = async (input: string | URL | Request): Promise => { + requests.push(input.toString()); + return Response.json({ + data: [ + { + id: "moonshotai/kimi-k2.7-code", + display_name: "Kimi K2.7 Code", + status: 1, + context_size: 262144, + max_output_tokens: 131072, + input_token_price_per_m: 9500, + output_token_price_per_m: 40000, + pricing: { + input_cache_read: { + price_per_m: 1900, + }, + }, + features: ["serverless", "function-calling", "structured-outputs", "reasoning"], + endpoints: ["chat/completions", "anthropic"], + input_modalities: ["text", "image", "video"], + }, + { + id: "qwen/qwen3-8b-fp8", + status: 4, + context_size: 128000, + max_output_tokens: 20000, + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + { + id: "ai_infer_test_1", + status: 1, + context_size: 200000, + max_output_tokens: 200000, + features: ["function-calling"], + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + ], + }); + }; + + const options = novitaModelManagerOptions({ fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + const model = models?.find(item => item.id === "moonshotai/kimi-k2.7-code"); + + expect(requests).toEqual(["https://api.novita.ai/openai/v1/models"]); + expect(models?.map(item => item.id)).toEqual(["moonshotai/kimi-k2.7-code"]); + expect(model?.provider).toBe("novita"); + expect(model?.baseUrl).toBe("https://api.novita.ai/openai/v1"); + expect(model?.name).toBe("Kimi K2.7 Code"); + expect(model?.reasoning).toBe(true); + expect(model?.supportsTools).toBe(true); + expect(model?.input).toEqual(["text", "image"]); + expect(model?.cost).toEqual({ input: 9.5, output: 40, cacheRead: 1.9, cacheWrite: 0 }); + expect(model?.contextWindow).toBe(262144); + expect(model?.maxTokens).toBe(131072); + }); +}); From 2dafa7ac79c59010a0fab01c82fd5ddc6fbfab88 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 11:32:04 +0200 Subject: [PATCH 051/205] feat: further codex metadata --- .../src/compaction/compaction-v2-streaming.ts | 45 +- packages/agent/src/compaction/compaction.ts | 54 ++- packages/agent/src/compaction/openai.ts | 39 +- packages/agent/test/remote-compaction.test.ts | 343 +++++++++++++- packages/ai/CHANGELOG.md | 1 + .../src/providers/openai-codex-responses.ts | 426 +++++++++++++++++- packages/ai/src/providers/openai-shared.ts | 52 ++- packages/ai/src/stream.ts | 1 + packages/ai/src/types.ts | 26 ++ packages/ai/test/issue-1701-repro.test.ts | 13 +- .../test/openai-codex-responses-lite.test.ts | 262 ++++++++++- packages/ai/test/openai-codex-stream.test.ts | 359 ++++++++++++++- .../openai-responses-history-payload.test.ts | 13 +- .../openai-responses-stream-terminal.test.ts | 10 +- packages/catalog/src/models.json | 10 +- packages/catalog/src/wire/codex.ts | 7 + .../coding-agent/src/session/agent-session.ts | 74 ++- .../agent-session-eager-compaction.test.ts | 35 +- 18 files changed, 1658 insertions(+), 112 deletions(-) diff --git a/packages/agent/src/compaction/compaction-v2-streaming.ts b/packages/agent/src/compaction/compaction-v2-streaming.ts index 251573775..14a2318f8 100644 --- a/packages/agent/src/compaction/compaction-v2-streaming.ts +++ b/packages/agent/src/compaction/compaction-v2-streaming.ts @@ -7,9 +7,14 @@ * compaction item as replacement history. */ -import type { Api, FetchImpl, Model } from "@oh-my-pi/pi-ai"; +import type { Api, CodexCompactionContext, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai"; import { isTransientStatus, ProviderHttpError } from "@oh-my-pi/pi-ai/error"; import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; +import { + createOpenAICodexCompactionRequestContext, + createOpenAICodexCompatibilityMetadata, + type OpenAICodexCompatibilityMetadata, +} from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { getOpenAIPromptCacheKey, getOpenAIResponsesRoutingSessionId, @@ -220,6 +225,8 @@ export async function requestCompactionV2Streaming( fetch?: FetchImpl; timeoutMs?: number; retryWait?: (delayMs: number, signal?: AbortSignal) => Promise; + providerSessionState?: Map; + codexCompaction?: CodexCompactionContext; }, ): Promise { const endpoint = getCompactionV2Endpoint(model); @@ -229,12 +236,32 @@ export async function requestCompactionV2Streaming( const fetchImpl = options?.fetch ?? globalThis.fetch; const retryWait = options?.retryWait ?? ((delayMs: number) => Bun.sleep(delayMs)); + const isCodexResponses = compactionV2Api(model) === "openai-codex-responses" || model.provider === "openai-codex"; + const codexMetadata = isCodexResponses + ? createOpenAICodexCompatibilityMetadata({ + sessionId: request.sessionId, + providerSessionState: options?.providerSessionState, + requestKind: "compaction", + compaction: createOpenAICodexCompactionRequestContext({ + context: options?.codexCompaction, + implementation: "responses_compaction_v2", + }), + }) + : undefined; let lastError: Error | undefined; for (let attempt = 0; attempt <= V2_COMPACTION_MAX_RETRIES; attempt++) { const timeoutSignal = withRequestTimeout(signal, options?.timeoutMs ?? V2_COMPACTION_TIMEOUT_MS); try { - return await attemptCompactionV2Streaming(endpoint, apiKey, model, request, fetchImpl, timeoutSignal); + return await attemptCompactionV2Streaming( + endpoint, + apiKey, + model, + request, + fetchImpl, + timeoutSignal, + codexMetadata, + ); } catch (err) { const error = err instanceof Error ? err : new Error(String(err)); if (signal?.aborted) throw error; @@ -265,6 +292,7 @@ async function attemptCompactionV2Streaming( request: CompactionV2Request, fetchImpl: FetchImpl, signal?: AbortSignal, + codexMetadata?: OpenAICodexCompatibilityMetadata, ): Promise { // Faithful to Codex: append the compaction trigger as the final input item // of an otherwise-normal Responses request, then stream the result. `store` @@ -287,6 +315,9 @@ async function attemptCompactionV2Streaming( ...(promptCacheKey ? { prompt_cache_key: promptCacheKey } : {}), ...(request.tools && request.tools.length > 0 ? { tools: request.tools, tool_choice: "auto" } : {}), }; + if (codexMetadata) { + body.client_metadata = codexMetadata.clientMetadata; + } // Responses Lite models take the same rewrite on the compaction stream: // instructions/tools ride as input items (codex-rs `compact_remote_v2` // builds through `build_responses_request`). @@ -295,7 +326,7 @@ async function attemptCompactionV2Streaming( } const response = await fetchImpl(endpoint, { method: "POST", - headers: buildCompactionV2Headers(model, apiKey, request), + headers: buildCompactionV2Headers(model, apiKey, request, codexMetadata), body: JSON.stringify(body), signal, }); @@ -320,7 +351,12 @@ async function attemptCompactionV2Streaming( return collectCompactionV2Output(response, request); } -function buildCompactionV2Headers(model: Model, apiKey: string, request: CompactionV2Request): Record { +function buildCompactionV2Headers( + model: Model, + apiKey: string, + request: CompactionV2Request, + codexMetadata?: OpenAICodexCompatibilityMetadata, +): Record { const api = compactionV2Api(model); const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey }; const routingSessionId = getOpenAIResponsesRoutingSessionId(cacheOptions); @@ -355,6 +391,7 @@ function buildCompactionV2Headers(model: Model, apiKey: string, request: Compact headers[OPENAI_HEADERS.RESPONSES_LITE] = "true"; } } + if (codexMetadata) Object.assign(headers, codexMetadata.headers); return headers; } diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 36b8feadd..772f0b232 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -9,18 +9,21 @@ import { type Api, type ApiKey, type AssistantMessage, + type CodexCompactionContext, type Context, Effort, type FetchImpl, type Message, type MessageAttribution, type Model, + type ProviderSessionState, type SimpleStreamOptions, type Tool, type Usage, withAuth, } from "@oh-my-pi/pi-ai"; import { ProviderHttpError } from "@oh-my-pi/pi-ai/error"; +import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses"; import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/providers/openai-shared"; import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; @@ -736,6 +739,10 @@ export interface SummaryOptions { sessionId?: string; /** Prompt-cache key for remote compaction transports that support provider prefix caching. */ promptCacheKey?: string; + /** Mutable provider state used to keep Codex compaction on the live session identity. */ + providerSessionState?: Map; + /** Classification shared by every provider request in this logical compaction. */ + codexCompaction?: CodexCompactionContext; /** Provider-visible tools for remote compaction transports that replay native tool history. */ tools?: Tool[]; /** Optional fetch implementation threaded into remote compaction calls. */ @@ -755,6 +762,13 @@ export interface SummaryOptions { ) => Promise; } +function localCodexCompaction(options: SummaryOptions | undefined) { + return createOpenAICodexCompactionRequestContext({ + context: options?.codexCompaction, + implementation: "responses", + }); +} + function formatPreviousSnapcompactArchive(archiveText: string): string { return prompt.render(snapcompactArchiveContextPrompt, { archiveText }); } @@ -844,6 +858,11 @@ export async function generateSummary( reasoning: resolveCompactionEffort(model, options?.thinkingLevel), initiatorOverride: options?.initiatorOverride, metadata: options?.metadata, + fetch: options?.fetch, + sessionId: options?.sessionId, + promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: localCodexCompaction(options), }, { telemetry: options?.telemetry, oneshotKind: "compaction_summary", completeImpl: options?.completeImpl }, ); @@ -1047,6 +1066,11 @@ async function generateShortSummary( reasoning: resolveCompactionEffort(model, options?.thinkingLevel), initiatorOverride: options?.initiatorOverride, metadata: options?.metadata, + fetch: options?.fetch, + sessionId: options?.sessionId, + promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: localCodexCompaction(options), }, { telemetry: options?.telemetry, oneshotKind: "compaction_short_summary", completeImpl: options?.completeImpl }, ); @@ -1317,6 +1341,8 @@ export async function compact( thinkingLevel: options?.thinkingLevel, sessionId: options?.sessionId, promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: options?.codexCompaction, tools: options?.tools, fetch: options?.fetch, completeImpl: options?.completeImpl, @@ -1375,7 +1401,12 @@ export async function compact( ); const remote = await withAuth( apiKey, - key => requestCompactionV2Streaming(model, key, request, signal, { fetch: summaryOptions.fetch }), + key => + requestCompactionV2Streaming(model, key, request, signal, { + fetch: summaryOptions.fetch, + providerSessionState: summaryOptions.providerSessionState, + codexCompaction: summaryOptions.codexCompaction, + }), { signal }, ); preserveData = { ...(preserveData ?? {}), ...storeCompactionV2PreserveData(remote, model) }; @@ -1419,7 +1450,12 @@ export async function compact( remoteHistory, summaryOptions.remoteInstructions ?? SUMMARIZATION_SYSTEM_PROMPT, signal, - { fetch: summaryOptions.fetch }, + { + fetch: summaryOptions.fetch, + sessionId: summaryOptions.sessionId, + providerSessionState: summaryOptions.providerSessionState, + codexCompaction: summaryOptions.codexCompaction, + }, ), { signal }, ); @@ -1495,16 +1531,9 @@ export async function compact( const shortSummary = usedRemoteCompaction ? "Remote compaction" : await generateShortSummary(recentMessages, summary, model, reserveTokens, apiKey, signal, { + ...summaryOptions, extraContext: options?.extraContext, - remoteEndpoint: summaryOptions.remoteEndpoint, - initiatorOverride: summaryOptions.initiatorOverride, - metadata: summaryOptions.metadata, - telemetry: summaryOptions.telemetry, - // Same propagation as summaryOptions above — generateShortSummary - // resolves its own reasoning via resolveCompactionEffort. thinkingLevel: options?.thinkingLevel, - fetch: summaryOptions.fetch, - completeImpl: summaryOptions.completeImpl, }); // Compute file lists and append to summary @@ -1567,6 +1596,11 @@ async function generateTurnPrefixSummary( reasoning: resolveCompactionEffort(model, options?.thinkingLevel), initiatorOverride: options?.initiatorOverride, metadata: options?.metadata, + fetch: options?.fetch, + sessionId: options?.sessionId, + promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: localCodexCompaction(options), }, { telemetry: options?.telemetry, oneshotKind: "compaction_turn_prefix", completeImpl: options?.completeImpl }, ); diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index aa4f37bb4..fbac0a8b7 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -17,9 +17,21 @@ import { ProviderHttpError } from "@oh-my-pi/pi-ai/error"; import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; +import { + createOpenAICodexCompactionRequestContext, + createOpenAICodexCompatibilityMetadata, +} from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { parseAzureDeploymentNameMap, parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; -import type { Api, AssistantMessage, FetchImpl, Message, Model } from "@oh-my-pi/pi-ai/types"; +import type { + Api, + AssistantMessage, + CodexCompactionContext, + FetchImpl, + Message, + Model, + ProviderSessionState, +} from "@oh-my-pi/pi-ai/types"; import { getOpenAIResponsesHistoryItems, getOpenAIResponsesHistoryPayload, @@ -461,7 +473,13 @@ export async function requestOpenAiRemoteCompaction( compactInput: Array>, instructions: string, signal?: AbortSignal, - opts?: { fetch?: FetchImpl; timeoutMs?: number }, + opts?: { + fetch?: FetchImpl; + timeoutMs?: number; + sessionId?: string; + providerSessionState?: Map; + codexCompaction?: CodexCompactionContext; + }, ): Promise { const endpoint = resolveOpenAiCompactEndpoint(model); const requestModel = resolveOpenAiCompactModel(model); @@ -474,6 +492,8 @@ export async function requestOpenAiRemoteCompaction( instructions, }; const isAzureOpenAiResponses = (model.remoteCompaction?.api ?? model.api) === "azure-openai-responses"; + const isCodexResponses = + model.provider === "openai-codex" || (model.remoteCompaction?.api ?? model.api) === "openai-codex-responses"; const headers: Record = isAzureOpenAiResponses ? { "content-type": "application/json", @@ -487,13 +507,26 @@ export async function requestOpenAiRemoteCompaction( }; // Codex endpoints require additional auth headers - if (model.provider === "openai-codex") { + if (isCodexResponses) { const accountId = getCodexAccountId(apiKey); if (accountId) { headers[OPENAI_HEADERS.ACCOUNT_ID] = accountId; } headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES; headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX; + Object.assign( + headers, + createOpenAICodexCompatibilityMetadata({ + sessionId: opts?.sessionId, + providerSessionState: opts?.providerSessionState, + requestKind: "compaction", + compaction: createOpenAICodexCompactionRequestContext({ + context: opts?.codexCompaction, + implementation: "responses_compact", + }), + includeInstallationHeader: true, + }).headers, + ); // Responses Lite models take the same rewrite on `/responses/compact`: // instructions ride as an input item and the lite marker header is set // (codex-rs routes compaction through `build_responses_request`). diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index 237c81557..78d4cb97b 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, test, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; import { type CompactionPreparation, compact, @@ -18,10 +18,36 @@ import { shouldUseOpenAiRemoteCompaction, } from "@oh-my-pi/pi-agent-core/compaction/openai"; import * as ai from "@oh-my-pi/pi-ai"; -import type { AssistantMessage, FetchImpl, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { getOpenAICodexTransportDetails } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import type { + AssistantMessage, + CodexCompactionContext, + FetchImpl, + Model, + ProviderSessionState, + ToolResultMessage, +} from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; -import { isRecord } from "@oh-my-pi/pi-utils"; +import * as piUtils from "@oh-my-pi/pi-utils"; + +const { isRecord } = piUtils; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; +const TEST_CODEX_COMPACTION: CodexCompactionContext = { + operationId: "compaction-operation-1", + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + strategy: "memento", +}; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); function makeOpenAiModel(overrides: Partial> = {}): Model<"openai-responses"> { return buildModel({ @@ -394,7 +420,9 @@ describe("requestCompactionV2Streaming", () => { }); describe("Responses Lite remote compaction", () => { - function makeCodexLiteModel(): Model<"openai-codex-responses"> { + function makeCodexLiteModel( + overrides: Partial> = {}, + ): Model<"openai-codex-responses"> { return buildModel({ id: "gpt-5.6-terra", name: "GPT-5.6 Terra", @@ -402,12 +430,14 @@ describe("Responses Lite remote compaction", () => { provider: "openai-codex", baseUrl: "https://chatgpt.example/backend-api", reasoning: true, + preferWebsockets: false, input: ["text", "image"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 372000, maxTokens: 128000, useResponsesLite: true, remoteCompaction: { enabled: true, api: "openai-codex-responses", v2StreamingEnabled: true }, + ...overrides, }); } @@ -415,22 +445,42 @@ describe("Responses Lite remote compaction", () => { instructions?: unknown; tools?: unknown; input?: Array>; + client_metadata?: unknown; } - function captureLite(init: RequestInit | undefined): { body: CapturedLiteRequest; liteHeader?: string } { + interface CapturedLiteExchange { + body: CapturedLiteRequest; + headers: Headers; + } + + function parseCodexTurnMetadata(value: unknown): Record { + if (typeof value !== "string") throw new Error("expected x-codex-turn-metadata"); + const parsed: unknown = JSON.parse(value); + if (!isRecord(parsed)) throw new Error("expected Codex turn metadata object"); + return parsed; + } + + function captureLite(init: RequestInit | undefined): CapturedLiteExchange { if (!init?.headers || init.headers instanceof Headers || Array.isArray(init.headers)) { throw new Error("Expected remote compaction to send headers as a plain object"); } - const rawLite = init.headers["x-openai-internal-codex-responses-lite"]; return { body: JSON.parse(String(init.body)) as CapturedLiteRequest, - liteHeader: typeof rawLite === "string" ? rawLite : undefined, + headers: new Headers(init.headers), + }; + } + + function captureStreamLite(init: RequestInit | undefined): CapturedLiteExchange { + if (!init?.headers) throw new Error("Expected local compaction request headers"); + return { + body: JSON.parse(String(init.body)) as CapturedLiteRequest, + headers: new Headers(init.headers), }; } test("V1 compaction sends the lite header and input-item instructions", async () => { const model = makeCodexLiteModel(); - let captured: { body: CapturedLiteRequest; liteHeader?: string } | undefined; + let captured: CapturedLiteExchange | undefined; const fetchMock: FetchImpl = async (_input, init) => { captured = captureLite(init); return Response.json({ output: [{ type: "compaction", encrypted_content: "enc" }] }); @@ -442,12 +492,29 @@ describe("Responses Lite remote compaction", () => { [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], "compact instructions", undefined, - { fetch: fetchMock }, + { + fetch: fetchMock, + sessionId: "codex-compaction-session", + providerSessionState: new Map(), + codexCompaction: TEST_CODEX_COMPACTION, + }, ); - expect(captured?.liteHeader).toBe("true"); + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); expect(captured?.body.instructions).toBeUndefined(); expect(captured?.body.tools).toBeUndefined(); + expect(captured?.body.client_metadata).toBeUndefined(); + expect(captured?.headers.get("x-codex-installation-id")).toBe(TEST_INSTALLATION_ID); + expect(captured?.headers.get("session-id")).toBe("codex-compaction-session"); + const v1TurnMetadata = parseCodexTurnMetadata(captured?.headers.get("x-codex-turn-metadata")); + expect(v1TurnMetadata.request_kind).toBe("compaction"); + expect(v1TurnMetadata.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses_compact", + phase: "pre_turn", + strategy: "memento", + }); expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); expect(captured?.body.input?.[1]).toEqual({ type: "message", @@ -462,8 +529,9 @@ describe("Responses Lite remote compaction", () => { model, [{ type: "message", role: "user", content: [{ type: "input_text", text: "real user" }] }], "compact instructions", + { sessionId: "codex-compaction-session" }, ); - let captured: { body: CapturedLiteRequest; liteHeader?: string } | undefined; + let captured: CapturedLiteExchange | undefined; const fetchMock: FetchImpl = async (_input, init) => { captured = captureLite(init); return sseResponse([ @@ -477,11 +545,30 @@ describe("Responses Lite remote compaction", () => { }; expect(shouldUseCompactionV2Streaming(model)).toBe(true); - await requestCompactionV2Streaming(model, "test-key", request, undefined, { fetch: fetchMock }); + await requestCompactionV2Streaming(model, "test-key", request, undefined, { + fetch: fetchMock, + providerSessionState: new Map(), + codexCompaction: TEST_CODEX_COMPACTION, + }); - expect(captured?.liteHeader).toBe("true"); + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); expect(captured?.body.instructions).toBeUndefined(); expect(captured?.body.tools).toBeUndefined(); + if (!isRecord(captured?.body.client_metadata)) throw new Error("expected V2 client_metadata"); + const v2ClientMetadata = captured.body.client_metadata; + const v2TurnMetadata = parseCodexTurnMetadata(v2ClientMetadata["x-codex-turn-metadata"]); + expect(captured.headers.get("x-codex-installation-id")).toBeNull(); + expect(v2ClientMetadata["x-codex-installation-id"]).toBe(TEST_INSTALLATION_ID); + expect(v2ClientMetadata.session_id).toBe(captured.headers.get("session-id")); + expect(v2ClientMetadata.thread_id).toBe(captured.headers.get("thread-id")); + expect(v2TurnMetadata.request_kind).toBe("compaction"); + expect(v2TurnMetadata.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses_compaction_v2", + phase: "pre_turn", + strategy: "memento", + }); expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); expect(captured?.body.input?.[1]).toEqual({ type: "message", @@ -490,6 +577,236 @@ describe("Responses Lite remote compaction", () => { }); expect(captured?.body.input?.at(-1)).toEqual({ type: "compaction_trigger" }); }); + + test("compact fan-out keeps local Codex summaries on one classified turn", async () => { + const model = makeCodexLiteModel(); + const captured: CapturedLiteExchange[] = []; + const fetchMock: FetchImpl = async (_input, init) => { + captured.push(captureStreamLite(init)); + return sseResponse([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "message", id: "msg_summary", role: "assistant", status: "in_progress", content: [] }, + }, + { + type: "response.content_part.added", + output_index: 0, + content_index: 0, + part: { type: "output_text", text: "" }, + }, + { type: "response.output_text.delta", output_index: 0, content_index: 0, delta: "local summary" }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "message", + id: "msg_summary", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "local summary" }], + }, + }, + { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 8, + output_tokens: 2, + total_tokens: 10, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]); + }; + const preparation: CompactionPreparation = { + firstKeptEntryId: "kept-1", + messagesToSummarize: [{ role: "user", content: "long history", timestamp: 1 }], + turnPrefixMessages: [], + recentMessages: [{ role: "user", content: "recent", timestamp: 2 }], + isSplitTurn: false, + tokensBefore: 100_000, + fileOps: createFileOps(), + settings: { + ...DEFAULT_COMPACTION_SETTINGS, + remoteEnabled: false, + remoteStreamingV2Enabled: false, + }, + }; + + const result = await compact(preparation, model, "test-key", undefined, undefined, { + fetch: fetchMock, + sessionId: "codex-compaction-session", + providerSessionState: new Map(), + codexCompaction: TEST_CODEX_COMPACTION, + }); + + expect(result.summary).toContain("local summary"); + expect(captured).toHaveLength(2); + const turnIds: string[] = []; + for (const exchange of captured) { + if (!isRecord(exchange.body.client_metadata)) throw new Error("expected local client_metadata"); + const clientMetadata = exchange.body.client_metadata; + const turnMetadata = parseCodexTurnMetadata(clientMetadata["x-codex-turn-metadata"]); + expect(exchange.headers.get("x-codex-installation-id")).toBeNull(); + expect(clientMetadata["x-codex-installation-id"]).toBe(TEST_INSTALLATION_ID); + expect(turnMetadata.request_kind).toBe("compaction"); + expect(turnMetadata.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }); + if (typeof turnMetadata.turn_id !== "string") throw new Error("expected Codex turn id"); + turnIds.push(turnMetadata.turn_id); + } + expect(new Set(turnIds).size).toBe(1); + }); + + test("local Codex compaction isolates and closes transient websocket sessions", async () => { + const originalWebSocket = global.WebSocket; + const sockets: AgentCompactionWebSocket[] = []; + let responseCount = 0; + + class AgentCompactionWebSocket { + static readonly CONNECTING = 0; + static readonly OPEN = 1; + static readonly CLOSING = 2; + static readonly CLOSED = 3; + + readyState = AgentCompactionWebSocket.CONNECTING; + binaryType: "blob" | "arraybuffer" | "nodebuffer" = "blob"; + onopen: ((event: Event) => void) | null = null; + onmessage: ((event: MessageEvent) => void) | null = null; + onerror: ((event: Event) => void) | null = null; + onclose: ((event: Event) => void) | null = null; + readonly handshakeHeaders = { + "x-codex-turn-state": `agent-compaction-state-${sockets.length}`, + }; + + constructor( + readonly url: string, + readonly options?: { headers?: Record }, + ) { + sockets.push(this); + queueMicrotask(() => { + this.readyState = AgentCompactionWebSocket.OPEN; + this.onopen?.(new Event("open")); + }); + } + + send(_data: string): void { + responseCount += 1; + const responseId = `response-${responseCount}`; + const messageId = `message-${responseCount}`; + const text = sockets[0] === this ? "main response" : "local summary"; + const events: Record[] = [ + { + type: "response.output_item.added", + item: { type: "message", id: messageId, role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", delta: text }, + { + type: "response.output_item.done", + item: { + type: "message", + id: messageId, + role: "assistant", + status: "completed", + content: [{ type: "output_text", text }], + }, + }, + { + type: "response.done", + response: { + id: responseId, + status: "completed", + usage: { + input_tokens: 8, + output_tokens: 2, + total_tokens: 10, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + for (const event of events) { + this.onmessage?.({ data: JSON.stringify(event) } as MessageEvent); + } + } + + close(): void { + this.readyState = AgentCompactionWebSocket.CLOSED; + } + } + + const providerSessionState = new Map(); + try { + global.WebSocket = AgentCompactionWebSocket as unknown as typeof WebSocket; + const model = makeCodexLiteModel({ preferWebsockets: true }); + const sessionId = "agent-compaction-isolation"; + const fetchMock: FetchImpl = async () => { + throw new Error("Codex websocket compaction unexpectedly used SSE"); + }; + const main = await ai + .streamSimple( + model, + { + systemPrompt: ["You are a helpful assistant."], + messages: [{ role: "user", content: "Start the turn", timestamp: Date.now() }], + }, + { apiKey: "test-key", fetch: fetchMock, sessionId, providerSessionState }, + ) + .result(); + expect(main.stopReason).toBe("stop"); + expect(sockets).toHaveLength(1); + expect(sockets[0]?.readyState).toBe(AgentCompactionWebSocket.OPEN); + + const preparation: CompactionPreparation = { + firstKeptEntryId: "kept-1", + messagesToSummarize: [{ role: "user", content: "long history", timestamp: 1 }], + turnPrefixMessages: [], + recentMessages: [{ role: "user", content: "recent", timestamp: 2 }], + isSplitTurn: false, + tokensBefore: 100_000, + fileOps: createFileOps(), + settings: { + ...DEFAULT_COMPACTION_SETTINGS, + remoteEnabled: false, + remoteStreamingV2Enabled: false, + }, + }; + const result = await compact(preparation, model, "test-key", undefined, undefined, { + fetch: fetchMock, + sessionId, + providerSessionState, + codexCompaction: TEST_CODEX_COMPACTION, + }); + + expect(result.summary).toContain("local summary"); + expect(sockets).toHaveLength(3); + expect(sockets[0]?.readyState).toBe(AgentCompactionWebSocket.OPEN); + expect(sockets[1]?.readyState).toBe(AgentCompactionWebSocket.CLOSED); + expect(sockets[2]?.readyState).toBe(AgentCompactionWebSocket.CLOSED); + expect( + getOpenAICodexTransportDetails(model, { + sessionId, + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + hasTurnState: true, + }); + } finally { + for (const state of providerSessionState.values()) state.close(); + providerSessionState.clear(); + global.WebSocket = originalWebSocket; + } + }); }); test("uses configured OpenAI-compatible compaction for custom providers", async () => { diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index cb7865ac7..05e484348 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -20,6 +20,7 @@ ### Fixed - Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract +- Fixed sequential-cutoff Codex reasoning summaries repeating earlier content when atomic summary snapshots are replayed or extended. ## [16.3.15] - 2026-07-09 diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 86ca0085c..609a570fb 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -12,6 +12,7 @@ import { $flag, asRecord, fetchWithRetry, + getInstallId, logger, parseStreamingJson, readSseJson, @@ -24,6 +25,8 @@ import { getEnvApiKey } from "../stream"; import type { Api, AssistantMessage, + CodexCompactionContext, + CodexCompactionRequestContext, Context, FetchImpl, Model, @@ -128,9 +131,9 @@ export interface OpenAICodexResponsesOptions extends StreamOptions { */ responsesLite?: boolean; /** - * Extra `client_metadata` to include in the request body on both transports. - * The canonical Codex envelope is `client_metadata["x-codex-turn-metadata"]` - * (JSON string of thread/turn identifiers); flat keys are also accepted. + * Additional fields embedded in the canonical + * `client_metadata["x-codex-turn-metadata"]` JSON blob. Reserved identity + * keys are ignored; extras are never emitted as top-level metadata fields. */ clientMetadata?: Record; /** @@ -142,6 +145,49 @@ export interface OpenAICodexResponsesOptions extends StreamOptions { onModerationMetadata?: (metadata: unknown) => void; } +/** Inputs for synthesizing Codex request identity outside the normal stream path. */ +export interface OpenAICodexCompatibilityMetadataOptions { + sessionId?: string; + providerSessionState?: Map; + requestKind: OpenAICodexRequestKind; + compaction?: CodexCompactionRequestContext; + startNewTurn?: boolean; + turnStartedAtUnixMs?: number; + clientMetadata?: Readonly>; + /** Add the direct installation header required by `/responses/compact`. */ + includeInstallationHeader?: boolean; +} + +/** Canonical Codex body metadata and compatibility headers for one request. */ +export interface OpenAICodexCompatibilityMetadata { + clientMetadata: Record; + headers: Record; +} + +/** Live Codex session state to preserve after a successful history rewrite. */ +export interface OpenAICodexCompactionResetOptions { + providerSessionState?: Map; + sessionId?: string; + compaction: CodexCompactionContext; +} + +/** Add the selected wire implementation to one logical compaction context. */ +export function createOpenAICodexCompactionRequestContext(options: { + context: CodexCompactionContext | undefined; + implementation: "responses" | "responses_compaction_v2" | "responses_compact"; +}): CodexCompactionRequestContext | undefined { + const context = options.context; + if (!context) return undefined; + return { + operationId: context.operationId, + trigger: context.trigger, + reason: context.reason, + implementation: options.implementation, + phase: context.phase, + strategy: context.strategy, + }; +} + const CODEX_DEBUG = $flag("PI_CODEX_DEBUG"); const CODEX_MAX_RETRIES = 5; const CODEX_RETRY_DELAY_MS = 500; @@ -345,6 +391,237 @@ type CodexWebSocketSessionState = { interface CodexProviderSessionState extends ProviderSessionState { webSocketSessions: Map; webSocketPublicToPrivate: Map; + metadataSessions: Map; +} + +/** Request classification encoded in Codex turn metadata. */ +export type OpenAICodexRequestKind = "turn" | "prewarm" | "compaction"; + +interface CodexMetadataSessionState { + sessionId: string; + threadId: string; + windowId: string; + turnId?: string; + turnStartedAtUnixMs?: number; + compactionOperationId?: string; + reuseTurnForNextRequest?: boolean; +} + +interface CodexCompatibilityIdentity { + installationId: string; + sessionId: string; + threadId: string; + windowId: string; + turnMetadataJson?: string; +} + +interface CodexRequestMetadata extends CodexCompatibilityIdentity { + turnId: string; + turnMetadataJson: string; + clientMetadata: Record; +} + +const CODEX_RESERVED_METADATA_KEYS: Record = { + installation_id: true, + [OPENAI_HEADERS.INSTALLATION_ID]: true, + session_id: true, + thread_id: true, + turn_id: true, + window_id: true, + [OPENAI_HEADERS.WINDOW_ID]: true, + [OPENAI_HEADERS.TURN_METADATA]: true, + [OPENAI_HEADERS.PARENT_THREAD_ID]: true, + [OPENAI_HEADERS.SUBAGENT]: true, + request_kind: true, + compaction: true, + turn_started_at_unix_ms: true, + forked_from_thread_id: true, + parent_thread_id: true, + subagent_kind: true, + thread_source: true, + sandbox: true, + workspaces: true, +}; + +function createCodexMetadataSessionState(sessionId: string): CodexMetadataSessionState { + return { + sessionId, + threadId: crypto.randomUUID(), + windowId: crypto.randomUUID(), + }; +} + +function getOrCreateCodexMetadataSessionState( + sessionId: string, + providerState: CodexProviderSessionState | undefined, +): CodexMetadataSessionState { + if (!providerState) return createCodexMetadataSessionState(sessionId); + const existing = providerState.metadataSessions.get(sessionId); + if (existing) return existing; + const created = createCodexMetadataSessionState(sessionId); + providerState.metadataSessions.set(sessionId, created); + return created; +} + +function createCodexCompatibilityIdentity(session: CodexMetadataSessionState): CodexCompatibilityIdentity { + return { + installationId: getInstallId(), + sessionId: session.sessionId, + threadId: session.threadId, + windowId: session.windowId, + }; +} + +function resolveCodexStartNewTurn( + session: CodexMetadataSessionState, + requestKind: OpenAICodexRequestKind, + compaction: CodexCompactionRequestContext | undefined, + override: boolean | undefined, +): boolean { + if (requestKind !== "compaction") { + if (requestKind === "turn") { + const reuseCompactionTurn = session.reuseTurnForNextRequest === true; + session.reuseTurnForNextRequest = false; + session.compactionOperationId = undefined; + if (reuseCompactionTurn) return false; + } + return override ?? requestKind === "turn"; + } + if (!compaction) return override ?? false; + const startsNewOperation = session.compactionOperationId !== compaction.operationId; + if (startsNewOperation) session.reuseTurnForNextRequest = false; + session.compactionOperationId = compaction.operationId; + return override ?? (compaction.phase !== "mid_turn" && startsNewOperation); +} + +function toAsciiJsonString(value: Record): string { + return JSON.stringify(value).replace( + /[\x7f-\uffff]/g, + char => `\\u${char.charCodeAt(0).toString(16).padStart(4, "0")}`, + ); +} + +function createCodexRequestMetadata( + session: CodexMetadataSessionState, + requestKind: OpenAICodexRequestKind, + options: { + startNewTurn: boolean; + turnStartedAtUnixMs?: number; + clientMetadata?: Readonly>; + compaction?: CodexCompactionRequestContext; + }, +): CodexRequestMetadata { + if (options.startNewTurn || !session.turnId) { + session.turnId = crypto.randomUUID(); + session.turnStartedAtUnixMs = options.turnStartedAtUnixMs; + } + const identity = createCodexCompatibilityIdentity(session); + const extra: Record = {}; + const callerMetadata = options.clientMetadata; + if (callerMetadata) { + for (const key in callerMetadata) { + if (!CODEX_RESERVED_METADATA_KEYS[key]) extra[key] = callerMetadata[key]; + } + } + const turnMetadata: Record = { + installation_id: identity.installationId, + session_id: identity.sessionId, + thread_id: identity.threadId, + turn_id: session.turnId, + window_id: identity.windowId, + request_kind: requestKind, + }; + if (options.compaction) { + turnMetadata.compaction = { + trigger: options.compaction.trigger, + reason: options.compaction.reason, + implementation: options.compaction.implementation, + phase: options.compaction.phase, + strategy: options.compaction.strategy, + }; + } + if (session.turnStartedAtUnixMs !== undefined) { + turnMetadata.turn_started_at_unix_ms = session.turnStartedAtUnixMs; + } + for (const key in extra) turnMetadata[key] = extra[key]; + const turnMetadataJson = toAsciiJsonString(turnMetadata); + return { + ...identity, + turnId: session.turnId, + turnMetadataJson, + clientMetadata: { + [OPENAI_HEADERS.INSTALLATION_ID]: identity.installationId, + session_id: identity.sessionId, + thread_id: identity.threadId, + [OPENAI_HEADERS.WINDOW_ID]: identity.windowId, + turn_id: session.turnId, + [OPENAI_HEADERS.TURN_METADATA]: turnMetadataJson, + }, + }; +} + +function applyCodexCompatibilityHeaders(headers: Headers, metadata: CodexCompatibilityIdentity): void { + headers.set(OPENAI_HEADERS.SCOPED_SESSION_ID, metadata.sessionId); + headers.set(OPENAI_HEADERS.THREAD_ID, metadata.threadId); + headers.set(OPENAI_HEADERS.WINDOW_ID, metadata.windowId); + if (metadata.turnMetadataJson) { + headers.set(OPENAI_HEADERS.TURN_METADATA, metadata.turnMetadataJson); + } else { + headers.delete(OPENAI_HEADERS.TURN_METADATA); + } +} + +/** + * Synthesize Codex request identity for raw provider routes such as remote + * compaction while reusing the live session's thread, window, and turn. + */ +export function createOpenAICodexCompatibilityMetadata( + options: OpenAICodexCompatibilityMetadataOptions, +): OpenAICodexCompatibilityMetadata { + const providerState = getCodexProviderSessionState(options.providerSessionState); + const sessionId = normalizeOpenAIPromptCacheKey(options.sessionId) ?? crypto.randomUUID(); + const session = getOrCreateCodexMetadataSessionState(sessionId, providerState); + const startNewTurn = resolveCodexStartNewTurn( + session, + options.requestKind, + options.compaction, + options.startNewTurn, + ); + const metadata = createCodexRequestMetadata(session, options.requestKind, { + startNewTurn, + turnStartedAtUnixMs: options.turnStartedAtUnixMs ?? (startNewTurn || !session.turnId ? Date.now() : undefined), + clientMetadata: options.clientMetadata, + compaction: options.compaction, + }); + const headers = new Headers(); + applyCodexCompatibilityHeaders(headers, metadata); + if (options.includeInstallationHeader) { + headers.set(OPENAI_HEADERS.INSTALLATION_ID, metadata.installationId); + } + return { + clientMetadata: { ...metadata.clientMetadata }, + headers: Object.fromEntries(headers.entries()), + }; +} + +/** + * Invalidate Codex history-dependent transport state after compaction while + * retaining the session identity and live connection. + */ +export function resetOpenAICodexHistoryAfterCompaction(options: OpenAICodexCompactionResetOptions): void { + const providerState = options.providerSessionState?.get(CODEX_PROVIDER_SESSION_STATE_KEY); + if (!isCodexProviderSessionState(providerState)) return; + for (const websocketState of providerState.webSocketSessions.values()) { + resetCodexWebSocketAppendState(websocketState); + if (options.compaction.phase !== "mid_turn") websocketState.turnState = undefined; + } + const sessionId = normalizeOpenAIPromptCacheKey(options.sessionId); + if (!sessionId) return; + const metadataSession = providerState.metadataSessions.get(sessionId); + if (!metadataSession) return; + metadataSession.windowId = crypto.randomUUID(); + metadataSession.compactionOperationId = undefined; + metadataSession.reuseTurnForNextRequest = options.compaction.phase !== "standalone_turn"; } interface CodexRequestContext { @@ -355,8 +632,10 @@ interface CodexRequestContext { requestHeaders: Record; transportSessionId?: string; providerSessionState?: CodexProviderSessionState; + isolatedTransportState?: CodexProviderSessionState; websocketState?: CodexWebSocketSessionState; responsesLite: boolean; + requestMetadata?: CodexRequestMetadata; transformedBody: RequestBody; rawRequestDump: RawHttpRequestDump; } @@ -613,23 +892,37 @@ function createCodexProviderSessionState(): CodexProviderSessionState { const state: CodexProviderSessionState = { webSocketSessions: new Map(), webSocketPublicToPrivate: new Map(), + metadataSessions: new Map(), close: () => { for (const session of state.webSocketSessions.values()) { session.connection?.close("session_disposed"); } state.webSocketSessions.clear(); state.webSocketPublicToPrivate.clear(); + state.metadataSessions.clear(); }, }; return state; } +function isCodexProviderSessionState(state: ProviderSessionState | undefined): state is CodexProviderSessionState { + return ( + state !== undefined && + "webSocketSessions" in state && + state.webSocketSessions instanceof Map && + "webSocketPublicToPrivate" in state && + state.webSocketPublicToPrivate instanceof Map && + "metadataSessions" in state && + state.metadataSessions instanceof Map + ); +} + function getCodexProviderSessionState( providerSessionState: Map | undefined, ): CodexProviderSessionState | undefined { if (!providerSessionState) return undefined; - const existing = providerSessionState.get(CODEX_PROVIDER_SESSION_STATE_KEY) as CodexProviderSessionState | undefined; - if (existing) return existing; + const existing = providerSessionState.get(CODEX_PROVIDER_SESSION_STATE_KEY); + if (isCodexProviderSessionState(existing)) return existing; const created = createCodexProviderSessionState(); providerSessionState.set(CODEX_PROVIDER_SESSION_STATE_KEY, created); return created; @@ -915,19 +1208,56 @@ async function buildCodexRequestContext( }; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); + const isolatedTransportState = options?.codexCompaction ? createCodexProviderSessionState() : undefined; + const transportProviderSessionState = isolatedTransportState ?? providerSessionState; const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite); const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite); const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined; if (sessionKey && publicSessionKey) { - providerSessionState?.webSocketPublicToPrivate.set(publicSessionKey, sessionKey); + transportProviderSessionState?.webSocketPublicToPrivate.set(publicSessionKey, sessionKey); } + const sharedWebsocketState = + sessionKey && providerSessionState + ? isolatedTransportState + ? providerSessionState.webSocketSessions.get(sessionKey) + : getCodexWebSocketSessionState(sessionKey, providerSessionState) + : undefined; const websocketState = - sessionKey && providerSessionState ? getCodexWebSocketSessionState(sessionKey, providerSessionState) : undefined; - if (websocketState && !isCodexWithinTurnContinuation(context)) { - // codex-rs scopes `x-codex-turn-state` to a single user turn: tool-loop - // follow-ups echo it, a new user turn starts without it. + sessionKey && isolatedTransportState + ? getCodexWebSocketSessionState(sessionKey, isolatedTransportState) + : sharedWebsocketState; + if (isolatedTransportState && websocketState && sharedWebsocketState) { + websocketState.disableWebsocket = sharedWebsocketState.disableWebsocket; + websocketState.turnState = sharedWebsocketState.turnState; + websocketState.modelsEtag = sharedWebsocketState.modelsEtag; + } + const withinTurnContinuation = isCodexWithinTurnContinuation(context); + const metadataSessionId = transportSessionId ?? crypto.randomUUID(); + const metadataSession = getOrCreateCodexMetadataSessionState(metadataSessionId, providerSessionState); + const compaction = options?.codexCompaction; + const requestKind: OpenAICodexRequestKind = compaction ? "compaction" : "turn"; + const startNewTurn = resolveCodexStartNewTurn( + metadataSession, + requestKind, + compaction, + compaction ? undefined : !withinTurnContinuation, + ); + if (websocketState && startNewTurn) { + // Codex scopes turn-state to one turn. Mid-turn compaction and tool-loop + // follow-ups preserve it; new user or compaction turns start without it. websocketState.turnState = undefined; } + const requestMetadata = createCodexRequestMetadata(metadataSession, requestKind, { + startNewTurn, + turnStartedAtUnixMs: compaction + ? startNewTurn || !metadataSession.turnId + ? Date.now() + : undefined + : getCodexTurnStartedAtUnixMs(context), + clientMetadata: transformedBody.client_metadata, + compaction, + }); + transformedBody.client_metadata = requestMetadata.clientMetadata; return { apiKey, accountId, @@ -936,8 +1266,10 @@ async function buildCodexRequestContext( requestHeaders, transportSessionId, providerSessionState, + isolatedTransportState, websocketState, responsesLite, + requestMetadata, transformedBody, rawRequestDump, }; @@ -1065,19 +1397,21 @@ async function openCodexWebSocketTransport( }> { const canAppendBeforeRequest = websocketState.canAppend === true; const chainedBody = buildCodexChainedRequestBody(requestContext.transformedBody, websocketState); - // WebSocket frames cannot carry per-request HTTP headers, so the Responses - // Lite marker rides in `client_metadata` on every `response.create`. + // WebSocket frames cannot carry per-request HTTP headers. Canonical Codex + // request identity is already in `client_metadata`; connection-scoped + // compatibility values that can change after the upgrade ride alongside it + // on every `response.create`. + const websocketClientMetadata = { ...(chainedBody.client_metadata ?? {}) }; + if (requestContext.responsesLite) { + websocketClientMetadata[CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY] = "true"; + } + if (websocketState.turnState) { + websocketClientMetadata[X_CODEX_TURN_STATE_HEADER] = websocketState.turnState; + } let websocketRequest = { type: "response.create", ...chainedBody, - ...(requestContext.responsesLite - ? { - client_metadata: { - ...(chainedBody.client_metadata ?? {}), - [CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY]: "true", - }, - } - : {}), + client_metadata: websocketClientMetadata, }; const replacementWebsocketRequest = await options?.onPayload?.(websocketRequest, model); if (replacementWebsocketRequest !== undefined) { @@ -1092,6 +1426,7 @@ async function openCodexWebSocketTransport( "websocket", websocketState, requestContext.responsesLite, + requestContext.requestMetadata, ); const requestBodyForState = structuredCloneJSON(requestContext.transformedBody); // `onPayload` may rewrite the outgoing frame (e.g. drop `stream_options`); @@ -1137,6 +1472,16 @@ async function openCodexWebSocketTransport( }; } +function getCodexTurnStartedAtUnixMs(context: Context): number { + for (let i = context.messages.length - 1; i >= 0; i--) { + const message = context.messages[i]; + if (message?.role === "user" && Number.isFinite(message.timestamp)) { + return Math.trunc(message.timestamp); + } + } + return Date.now(); +} + /** * True when the request continues the current turn (everything after the * last assistant message is tool results), false when a new user turn starts. @@ -1177,6 +1522,7 @@ async function openCodexSseTransport( wireBody, state, requestContext.responsesLite, + requestContext.requestMetadata, requestSetup.requestSignal, requestSetup.firstEventTimeoutMs, event => options?.onSseEvent?.(event, model), @@ -1591,7 +1937,9 @@ class CodexStreamProcessor { const contentIndex = entry?.contentIndex ?? output.content.length - 1; if (item.type === "reasoning" && block?.type === "thinking") { - block.thinking = finalizeReasoningThinking(item, block.thinking); + block.thinking = finalizeReasoningThinking(item, block.thinking, { + cumulativeSummarySnapshots: this.#sequentialCutoffSummaries, + }); block.thinkingSignature = JSON.stringify(item); stream.push({ type: "thinking_end", @@ -2168,6 +2516,8 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses" stream.push({ type: "error", reason: "error", error: output }); } stream.end(); + } finally { + requestContext?.isolatedTransportState?.close(); } })(); @@ -2198,6 +2548,11 @@ export async function prewarmOpenAICodexResponses( if (!sessionKey || !providerSessionState) return; const state = getCodexWebSocketSessionState(sessionKey, providerSessionState); if (!shouldUseCodexWebSocket(model, state, options?.preferWebsockets)) return; + const metadataSession = getOrCreateCodexMetadataSessionState( + transportSessionId ?? crypto.randomUUID(), + providerSessionState, + ); + const requestIdentity = createCodexCompatibilityIdentity(metadataSession); const headers = logger.time( "prewarmCodex:createHeaders", createCodexHeaders, @@ -2208,6 +2563,7 @@ export async function prewarmOpenAICodexResponses( "websocket", state, responsesLite, + requestIdentity, ); await logger.time( "prewarmCodex:establishWs", @@ -2315,6 +2671,7 @@ export interface OpenAICodexTransportDetails { canAppend: boolean; prewarmed: boolean; hasSessionState: boolean; + hasTurnState: boolean; lastFallbackAt?: number; } @@ -2377,6 +2734,7 @@ export function getOpenAICodexTransportDetails( canAppend: state?.canAppend ?? false, prewarmed: state?.prewarmed ?? false, hasSessionState: state !== undefined, + hasTurnState: state?.turnState !== undefined, lastFallbackAt: state?.lastFallbackAt, }; } @@ -3320,12 +3678,22 @@ async function openCodexSseEventStream( body: RequestBody, state: CodexWebSocketSessionState | undefined, responsesLite: boolean, + requestMetadata: CodexRequestMetadata | undefined, signal: AbortSignal | undefined, firstEventTimeoutMs: number | undefined, onSseEvent?: OpenAICodexResponsesOptions["onSseEvent"], fetchOverride?: FetchImpl, ): Promise>> { - const headers = createCodexHeaders(requestHeaders, accountId, apiKey, sessionId, "sse", state, responsesLite); + const headers = createCodexHeaders( + requestHeaders, + accountId, + apiKey, + sessionId, + "sse", + state, + responsesLite, + requestMetadata, + ); CODEX_DEBUG && logger.debug("[codex] codex request", { url, @@ -3385,6 +3753,7 @@ function createCodexHeaders( transport: CodexTransport = "sse", state?: CodexWebSocketSessionState, responsesLite = false, + requestMetadata?: CodexCompatibilityIdentity, ): Headers { const headers = new Headers(initHeaders ?? {}); headers.delete("x-api-key"); @@ -3408,6 +3777,15 @@ function createCodexHeaders( headers.delete(OPENAI_HEADERS.SESSION_ID); headers.delete("x-client-request-id"); } + headers.delete(OPENAI_HEADERS.INSTALLATION_ID); + if (requestMetadata) { + applyCodexCompatibilityHeaders(headers, requestMetadata); + } else { + headers.delete(OPENAI_HEADERS.SCOPED_SESSION_ID); + headers.delete(OPENAI_HEADERS.THREAD_ID); + headers.delete(OPENAI_HEADERS.WINDOW_ID); + headers.delete(OPENAI_HEADERS.TURN_METADATA); + } if (state?.turnState) { headers.set(X_CODEX_TURN_STATE_HEADER, state.turnState); } else { @@ -3445,6 +3823,10 @@ function redactHeaders(headers: Headers): Record { lower.includes("account") || lower.includes("session") || lower.includes("conversation") || + lower.includes("thread") || + lower.includes("window") || + lower.includes("installation") || + lower.startsWith("x-codex-turn") || lower === "x-client-request-id" || lower === "cookie" ) { diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 1820c24f5..789f31213 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1702,9 +1702,36 @@ export function appendReasoningSummaryPart( item.summary.push(part); } -/** Chooses the final reasoning text without discarding content already streamed into the block. */ -export function finalizeReasoningThinking(item: ResponseReasoningItem, streamedThinking: string): string { - const summaryThinking = item.summary?.map(part => part.text).join("\n\n") ?? ""; +// Sequential-cutoff streams may repeat the full canonical summary as later parts. +function foldReasoningSummary(parts: ResponseReasoningItem["summary"] | undefined): string { + if (!parts) return ""; + let canonical = ""; + for (const part of parts) { + const text = part.text; + if (!text || text === canonical) continue; + const extendsCanonical = text.startsWith(canonical) && text[canonical.length] === "\n"; + canonical = !canonical || extendsCanonical ? text : `${canonical}\n\n${text}`; + } + return canonical; +} + +/** Chooses final reasoning text without making sequential-cutoff results disagree with emitted deltas. */ +export function finalizeReasoningThinking( + item: ResponseReasoningItem, + streamedThinking: string, + options: { cumulativeSummarySnapshots?: boolean } = {}, +): string { + const summaryThinking = options.cumulativeSummarySnapshots + ? foldReasoningSummary(item.summary) + : (item.summary?.map(part => part.text).join("\n\n") ?? ""); + if ( + options.cumulativeSummarySnapshots && + streamedThinking && + summaryThinking && + summaryThinking !== streamedThinking + ) { + return streamedThinking; + } if (summaryThinking) return summaryThinking; const contentThinking = item.content?.[0]?.type === "reasoning_text" ? (item.content[0].text ?? "") : ""; return contentThinking || streamedThinking || ""; @@ -1742,12 +1769,12 @@ export function appendReasoningSummaryPartDone( } /** - * Applies an atomic `response.reasoning_summary_text.done` event (concurrent - * reasoning summaries, `stream_options.reasoning_summary_delivery: - * "sequential_cutoff"`). The event carries the FULL text for `summaryIndex`; - * incremental `.delta`/`.part.*` events are ignored under this contract, so - * the whole part is stored and streamed here. Parts after the first are - * separated by a section break, mirroring codex-rs. + * Applies an atomic `response.reasoning_summary_text.done` snapshot. + * + * Sequential-cutoff streams can replay an index or send the accumulated + * summary as a later part. Rebuild the canonical summary and emit only its + * append-only suffix. Divergent corrections stay buffered until finalization + * so delta consumers never receive suffixes based on unseen replacement text. */ export function applyReasoningSummaryDone( item: ResponseReasoningItem, @@ -1763,8 +1790,11 @@ export function applyReasoningSummaryDone( item.summary.push({ type: "summary_text", text: "" }); } item.summary[summaryIndex].text = text; - const delta = summaryIndex > 0 ? `\n\n${text}` : text; - block.thinking += delta; + const after = foldReasoningSummary(item.summary); + if (!after.startsWith(block.thinking)) return; + const delta = after.slice(block.thinking.length); + if (!delta) return; + block.thinking = after; stream.push({ type: "thinking_delta", contentIndex, delta, partial: output }); } diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 02d17ad2f..de922d926 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1632,6 +1632,7 @@ function mapOptionsForApi( toolChoice: mapOpenAiToolChoice(options?.toolChoice), serviceTier: options?.serviceTier, preferWebsockets: options?.preferWebsockets, + codexCompaction: options?.codexCompaction, reasoningSummary: options?.hideThinkingSummary ? null : "detailed", textVerbosity: options?.textVerbosity, }); diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index e94884553..5b568de29 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -323,6 +323,30 @@ export interface RawSseEvent { raw: string[]; } +/** Lifecycle fields shared by every Codex compaction implementation. */ +export interface CodexCompactionContext { + /** Stable only for one logical compaction, including parallel summary calls. */ + operationId: string; + trigger: "manual" | "auto"; + reason: "user_requested" | "context_limit" | "model_downshift" | "comp_hash_changed"; + phase: "standalone_turn" | "pre_turn" | "mid_turn"; + strategy: "memento" | "prefix_compaction"; +} + +/** Canonical nested metadata serialized into the Codex turn envelope. */ +export interface CodexCompactionMetadata { + trigger: "manual" | "auto"; + reason: "user_requested" | "context_limit" | "model_downshift" | "comp_hash_changed"; + implementation: "responses" | "responses_compaction_v2" | "responses_compact"; + phase: "standalone_turn" | "pre_turn" | "mid_turn"; + strategy: "memento" | "prefix_compaction"; +} + +/** Dispatch context combining canonical metadata with its local operation identity. */ +export interface CodexCompactionRequestContext extends CodexCompactionMetadata { + operationId: string; +} + export interface StreamOptions { temperature?: number; topP?: number; @@ -398,6 +422,8 @@ export interface StreamOptions { * Providers can use this to persist transport/session state between turns. */ providerSessionState?: Map; + /** Canonical Codex compaction classification; ignored by other providers. */ + codexCompaction?: CodexCompactionRequestContext; /** * Force Gemini model-mode Interactions API transport for providers that support it. * When unset, those providers may still use Interactions to continue known diff --git a/packages/ai/test/issue-1701-repro.test.ts b/packages/ai/test/issue-1701-repro.test.ts index 4016d573d..79318532c 100644 --- a/packages/ai/test/issue-1701-repro.test.ts +++ b/packages/ai/test/issue-1701-repro.test.ts @@ -1,12 +1,23 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses"; import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import type { Context, Model, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import * as piUtils from "@oh-my-pi/pi-utils"; import { z } from "zod/v4"; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); + const completionsModel: Model<"openai-completions"> = buildModel({ id: "gpt-4o-mini-test", name: "GPT-4o Mini Test", diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index f70e57d43..169f578a9 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { type InputItem, type RequestBody, @@ -7,13 +7,25 @@ import { import { buildTransformedCodexRequestBody, convertCodexResponsesMessages, + resetOpenAICodexHistoryAfterCompaction, streamOpenAICodexResponses, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { isOpenAIResponsesProgressEvent } from "@oh-my-pi/pi-ai/providers/openai-shared"; -import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types"; +import type { CodexCompactionRequestContext, Context, FetchImpl, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import * as piUtils from "@oh-my-pi/pi-utils"; import { createCodexModel } from "./helpers"; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); + function createCodexTestToken(accountId = "acc_test"): string { const payload = Buffer.from( JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: accountId } }), @@ -64,6 +76,24 @@ interface CapturedCodexRequest { body: Record; } +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function requireRecord(value: unknown, label: string): Record { + if (!isRecord(value)) { + throw new Error(`expected ${label} to be an object`); + } + return value; +} + +function parseTurnMetadata(clientMetadata: Record): Record { + const encoded = clientMetadata["x-codex-turn-metadata"]; + if (typeof encoded !== "string") throw new Error("expected x-codex-turn-metadata"); + const decoded: unknown = JSON.parse(encoded); + return requireRecord(decoded, "x-codex-turn-metadata"); +} + function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRequest) => void): FetchImpl { return (async (input: string | URL, init?: RequestInit) => { const url = typeof input === "string" ? input : input.toString(); @@ -370,15 +400,21 @@ describe("openai-codex fresh execution input shaping", () => { }); describe("openai-codex Responses Lite and client metadata wire format", () => { - it("sends the lite header and client_metadata body field over SSE", async () => { + it("sends canonical Codex metadata and protects reserved fields over SSE", async () => { const model = createCodexModel("gpt-5.1-codex"); - const clientMetadata = { "x-codex-turn-metadata": '{"thread_id":"thread_1","turn_id":"turn_1"}' }; + const context = createCodexTestContext(); + const clientMetadata = { + workspace_kind: "repo", + workspace_path: "東京/🚀", + session_id: "caller-session", + "x-codex-turn-metadata": '{"turn_id":"caller-turn"}', + }; let captured: CapturedCodexRequest | undefined; const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { captured = request; }); - const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + const result = await streamOpenAICodexResponses(model, context, { apiKey: createCodexTestToken(), fetch: fetchMock, responsesLite: true, @@ -386,8 +422,141 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { }).result(); expect(result.stopReason).toBe("stop"); - expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); - expect(captured?.body.client_metadata).toEqual(clientMetadata); + if (!captured) throw new Error("expected a captured Codex request"); + expect(captured.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured.headers.get("x-codex-installation-id")).toBeNull(); + + const metadata = requireRecord(captured.body.client_metadata, "client_metadata"); + const turnMetadata = parseTurnMetadata(metadata); + expect(metadata.workspace_kind).toBeUndefined(); + expect(metadata.workspace_path).toBeUndefined(); + expect(metadata.session_id).not.toBe("caller-session"); + expect(turnMetadata.request_kind).toBe("turn"); + expect(turnMetadata.turn_started_at_unix_ms).toBe(context.messages[0]?.timestamp); + expect(turnMetadata.workspace_kind).toBe("repo"); + expect(turnMetadata.workspace_path).toBe("東京/🚀"); + expect(metadata["x-codex-installation-id"]).toMatch( + /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i, + ); + expect(metadata.session_id).toBe(turnMetadata.session_id); + expect(metadata.thread_id).toBe(turnMetadata.thread_id); + expect(metadata.turn_id).toBe(turnMetadata.turn_id); + expect(metadata["x-codex-window-id"]).toBe(turnMetadata.window_id); + expect(metadata.session_id).toBe(captured.headers.get("session-id")); + expect(metadata.thread_id).toBe(captured.headers.get("thread-id")); + expect(metadata["x-codex-window-id"]).toBe(captured.headers.get("x-codex-window-id")); + expect(metadata["x-codex-turn-metadata"]).toBe(captured.headers.get("x-codex-turn-metadata")); + const turnMetadataHeader = captured.headers.get("x-codex-turn-metadata"); + expect(turnMetadataHeader).toMatch(/^[\x20-\x7e]+$/); + const reparsedTurnMetadata: unknown = turnMetadataHeader ? JSON.parse(turnMetadataHeader) : undefined; + expect(requireRecord(reparsedTurnMetadata, "round-tripped turn metadata").workspace_path).toBe("東京/🚀"); + }); + + it("keeps the installation identity stable across provider sessions", async () => { + const model = createCodexModel("gpt-5.1-codex"); + const captured: CapturedCodexRequest[] = []; + const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { + captured.push(request); + }); + + await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + sessionId: "metadata-session-one", + }).result(); + await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + sessionId: "metadata-session-two", + }).result(); + + const firstMetadata = requireRecord(captured[0]?.body.client_metadata, "first client_metadata"); + const secondMetadata = requireRecord(captured[1]?.body.client_metadata, "second client_metadata"); + expect(firstMetadata["x-codex-installation-id"]).toBe(secondMetadata["x-codex-installation-id"]); + expect(firstMetadata.session_id).toBe("metadata-session-one"); + expect(secondMetadata.session_id).toBe("metadata-session-two"); + expect(firstMetadata.thread_id).not.toBe(secondMetadata.thread_id); + }); + + it("rotates compaction turns by phase and reuses one operation across fan-out calls", async () => { + const model = createCodexModel("gpt-5.1-codex"); + const providerSessionState = new Map(); + const captured: CapturedCodexRequest[] = []; + const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { + captured.push(request); + }); + const send = async (codexCompaction?: CodexCompactionRequestContext): Promise => { + await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + sessionId: "compaction-lifecycle-session", + providerSessionState, + codexCompaction, + }).result(); + }; + const preTurn: CodexCompactionRequestContext = { + operationId: "pre-turn-operation", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }; + const midTurn: CodexCompactionRequestContext = { + ...preTurn, + operationId: "mid-turn-operation", + phase: "mid_turn", + }; + const standalone: CodexCompactionRequestContext = { + ...preTurn, + operationId: "standalone-operation", + trigger: "manual", + reason: "user_requested", + phase: "standalone_turn", + }; + + await send(); + await send(preTurn); + await send(preTurn); + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId: "compaction-lifecycle-session", + compaction: preTurn, + }); + await send(); + await send(midTurn); + await send(standalone); + + const turns = captured.map((request, index) => + parseTurnMetadata(requireRecord(request.body.client_metadata, `client_metadata ${index}`)), + ); + expect(turns[0]?.request_kind).toBe("turn"); + expect(turns[1]?.turn_id).not.toBe(turns[0]?.turn_id); + expect(turns[2]?.turn_id).toBe(turns[1]?.turn_id); + expect(turns[2]?.turn_started_at_unix_ms).toBe(turns[1]?.turn_started_at_unix_ms); + expect(turns[3]?.request_kind).toBe("turn"); + expect(turns[3]?.turn_id).toBe(turns[1]?.turn_id); + expect(turns[3]?.window_id).not.toBe(turns[2]?.window_id); + expect(turns[4]?.turn_id).toBe(turns[1]?.turn_id); + expect(turns[5]?.turn_id).not.toBe(turns[4]?.turn_id); + expect(turns[1]?.thread_id).toBe(turns[5]?.thread_id); + expect(turns[1]?.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }); + const nestedCompaction = requireRecord(turns[1]?.compaction, "nested compaction metadata"); + expect(nestedCompaction.operationId).toBeUndefined(); + expect(nestedCompaction.operation_id).toBeUndefined(); + expect(turns[5]?.compaction).toEqual({ + trigger: "manual", + reason: "user_requested", + implementation: "responses", + phase: "standalone_turn", + strategy: "memento", + }); }); it("keeps lite and strips image detail when a lite request contains images", async () => { const model = buildModel({ @@ -461,7 +630,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect((captured?.body.input as Array>)[0]?.type).toBe("additional_tools"); }); - it("omits the lite header and client_metadata when not requested", async () => { + it("omits the lite marker while retaining canonical client_metadata", async () => { const model = createCodexModel("gpt-5.1-codex"); let captured: CapturedCodexRequest | undefined; const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { @@ -475,7 +644,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(result.stopReason).toBe("stop"); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBeNull(); - expect(captured?.body.client_metadata).toBeUndefined(); + expect(captured?.body.client_metadata).toBeDefined(); }); }); @@ -559,7 +728,7 @@ describe("openai-codex concurrent reasoning summaries", () => { expect(unsupported.stream_options).toBeUndefined(); }); - it("decodes atomic summary dones and ignores legacy deltas under sequential cutoff", async () => { + it("deduplicates cumulative atomic summaries and ignores legacy deltas under sequential cutoff", async () => { const model = createCodexModel("gpt-5.6-terra"); const events: Array> = [ { @@ -586,14 +755,70 @@ describe("openai-codex concurrent reasoning summaries", () => { item_id: "reason_1", output_index: 0, summary_index: 0, - text: "First part", + text: "Plan", }, { type: "response.reasoning_summary_text.done", item_id: "reason_1", output_index: 0, summary_index: 1, - text: "Second part", + text: "Planning details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 1, + text: "Planning details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nInspect", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nInspect details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nInspect details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 3, + text: "Plan\n\nPlanning details\n\nInspect details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nReview", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nReview output", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 3, + text: "Plan\n\nPlanning details\n\nReview output", }, { type: "response.output_item.done", @@ -602,8 +827,10 @@ describe("openai-codex concurrent reasoning summaries", () => { type: "reasoning", id: "reason_1", summary: [ - { type: "summary_text", text: "First part" }, - { type: "summary_text", text: "Second part" }, + { type: "summary_text", text: "Plan" }, + { type: "summary_text", text: "Planning details" }, + { type: "summary_text", text: "Plan\n\nPlanning details\n\nInspect details\n\nUnseen final" }, + { type: "summary_text", text: "Plan\n\nPlanning details\n\nInspect details\n\nUnseen final" }, ], }, }, @@ -618,7 +845,7 @@ describe("openai-codex concurrent reasoning summaries", () => { type: "response.reasoning_summary_text.done", item_id: "reason_1", output_index: 0, - summary_index: 2, + summary_index: 4, text: "STALE", }, { @@ -662,10 +889,11 @@ describe("openai-codex concurrent reasoning summaries", () => { const result = await stream.result(); expect(captured?.body.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); - expect(thinkingDeltas).toEqual(["First part", "\n\nSecond part"]); + expect(thinkingDeltas).toEqual(["Plan", "\n\nPlanning details", "\n\nInspect", " details"]); expect(result.stopReason).toBe("stop"); const thinking = result.content.find(block => block.type === "thinking"); - expect(thinking?.thinking).toBe("First part\n\nSecond part"); + expect(thinking?.thinking).toBe("Plan\n\nPlanning details\n\nInspect details"); + expect(thinking?.thinking).toBe(thinkingDeltas.join("")); const text = result.content.find(block => block.type === "text"); expect(text?.text).toBe("Hello"); }); diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 505039c5c..32b0b52ec 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -1,18 +1,29 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { streamSimple } from "@oh-my-pi/pi-ai"; import { getOpenAICodexTransportDetails, getOpenAICodexWebSocketDebugStats, prewarmOpenAICodexResponses, + resetOpenAICodexHistoryAfterCompaction, streamOpenAICodexResponses, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; -import type { Context, FetchImpl, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import type { + CodexCompactionRequestContext, + Context, + FetchImpl, + Model, + ModelSpec, + ProviderSessionState, +} from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import { getAgentDir, setAgentDir, TempDir } from "@oh-my-pi/pi-utils"; +import * as piUtils from "@oh-my-pi/pi-utils"; + +const { getAgentDir, setAgentDir, TempDir } = piUtils; const originalAgentDir = getAgentDir(); const originalWebSocket = global.WebSocket; const originalCodexWebSocketV2 = Bun.env.PI_CODEX_WEBSOCKET_V2; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; function restoreEnv(name: string, value: string | undefined): void { if (value === undefined) { @@ -22,6 +33,10 @@ function restoreEnv(name: string, value: string | undefined): void { Bun.env[name] = value; } +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + afterEach(() => { global.WebSocket = originalWebSocket; setAgentDir(originalAgentDir); @@ -60,6 +75,22 @@ function createCodexTestContext(): Context { }; } +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function requireRecord(value: unknown, label: string): Record { + if (!isRecord(value)) throw new Error(`expected ${label} to be an object`); + return value; +} + +function parseTurnMetadata(clientMetadata: Record): Record { + const encoded = clientMetadata["x-codex-turn-metadata"]; + if (typeof encoded !== "string") throw new Error("expected x-codex-turn-metadata"); + const decoded: unknown = JSON.parse(encoded); + return requireRecord(decoded, "x-codex-turn-metadata"); +} + function createCompletedCodexSse(text: string): string { return `${[ `data: ${JSON.stringify({ type: "response.content_part.added", part: { type: "output_text", text: "" } })}`, @@ -1222,7 +1253,7 @@ describe("openai-codex streaming", () => { sessionId: "ws-lite-session", providerSessionState: new Map(), responsesLite: true, - clientMetadata: { "x-codex-turn-metadata": '{"thread_id":"t_1"}' }, + clientMetadata: { workspace_kind: "repo", "x-codex-turn-metadata": '{"thread_id":"caller"}' }, }, ).result(); @@ -1230,10 +1261,28 @@ describe("openai-codex streaming", () => { expect(capturedHeaders?.["x-openai-internal-codex-responses-lite"]).toBe("true"); expect(sentRequests).toHaveLength(1); expect(sentRequests[0]?.type).toBe("response.create"); - expect(sentRequests[0]?.client_metadata).toEqual({ - "x-codex-turn-metadata": '{"thread_id":"t_1"}', + const metadata = requireRecord(sentRequests[0]?.client_metadata, "client_metadata"); + const turnMetadata = parseTurnMetadata(metadata); + expect(metadata).toMatchObject({ + session_id: "ws-lite-session", ws_request_header_x_openai_internal_codex_responses_lite: "true", + "x-codex-installation-id": TEST_INSTALLATION_ID, }); + expect(metadata.workspace_kind).toBeUndefined(); + expect(turnMetadata).toMatchObject({ + installation_id: TEST_INSTALLATION_ID, + session_id: "ws-lite-session", + thread_id: metadata.thread_id, + turn_id: metadata.turn_id, + window_id: metadata["x-codex-window-id"], + request_kind: "turn", + workspace_kind: "repo", + }); + expect(capturedHeaders?.["x-codex-installation-id"]).toBeUndefined(); + expect(metadata.session_id).toBe(capturedHeaders?.["session-id"]); + expect(metadata.thread_id).toBe(capturedHeaders?.["thread-id"]); + expect(metadata["x-codex-window-id"]).toBe(capturedHeaders?.["x-codex-window-id"]); + expect(metadata["x-codex-turn-metadata"]).toBe(capturedHeaders?.["x-codex-turn-metadata"]); }); it("streams SSE responses into AssistantMessageEventStream", async () => { @@ -2110,7 +2159,7 @@ describe("openai-codex streaming", () => { expect(fallbackDetails.fallbackCount).toBe(1); }); - it("immediately falls back to SSE on fatal websocket connection errors", async () => { + it("carries fatal websocket fallback into isolated compaction transport", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); @@ -2173,6 +2222,23 @@ describe("openai-codex streaming", () => { expect(result.role).toBe("assistant"); expect(constructorCount).toBe(1); expect(fetchMock).toHaveBeenCalledTimes(1); + const compacted = await streamOpenAICodexResponses(model, context, { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-fatal-fallback-session", + providerSessionState, + codexCompaction: { + operationId: "fallback-compaction", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }, + }).result(); + expect(compacted.stopReason).toBe("stop"); + expect(constructorCount).toBe(1); + expect(fetchMock).toHaveBeenCalledTimes(2); const transportDetails = getOpenAICodexTransportDetails(model, { sessionId: "ws-fatal-fallback-session", providerSessionState, @@ -2182,7 +2248,7 @@ describe("openai-codex streaming", () => { expect(transportDetails.fallbackCount).toBe(1); }); - it("captures websocket handshake metadata and replays it on later SSE requests", async () => { + it("isolates compaction transport and preserves main mid-turn state", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); @@ -2199,12 +2265,21 @@ describe("openai-codex streaming", () => { `data: ${JSON.stringify({ type: "response.output_item.done", item: { type: "message", id: "msg_sse", role: "assistant", status: "completed", content: [{ type: "output_text", text: "Hello SSE" }] } })}`, `data: ${JSON.stringify({ type: "response.completed", response: { status: "completed", usage: { input_tokens: 5, output_tokens: 3, total_tokens: 8, input_tokens_details: { cached_tokens: 0 } } } })}`, ].join("\n\n")}\n\n`; + let firstRequest: Record | undefined; + let continuationRequest: Record | undefined; + let continuationHeaders: Headers | undefined; const fetchMock = vi.fn(async (_input: string | URL, init?: RequestInit) => { - const headers = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers); - expect(headers.get("x-codex-turn-state")).toBe("ws-turn-state-1"); - expect(headers.get("x-models-etag")).toBe("models-etag-1"); + continuationHeaders = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers); + expect(continuationHeaders.get("x-codex-turn-state")).toBe("ws-turn-state-1"); + expect(continuationHeaders.get("x-models-etag")).toBe("models-etag-1"); + if (typeof init?.body !== "string") throw new Error("expected an SSE request body"); + const body: unknown = JSON.parse(init.body); + continuationRequest = requireRecord(body, "SSE continuation request"); return new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); }); + let websocketRequestCount = 0; + let websocketConstructorCount = 0; + const websocketInstances: MockWebSocket[] = []; class HandshakeWebSocket extends MockWebSocket { handshakeHeaders = { @@ -2215,11 +2290,29 @@ describe("openai-codex streaming", () => { constructor(url: string, options?: { headers?: WsHeaders }) { super(url, options); + websocketConstructorCount += 1; + websocketInstances.push(this); this.scheduleOpen(); } - send(): void { - this.emitCodexResponse({ messageId: "msg_ws", responseId: "resp_ws", text: "Hello WS" }); + send(data: string): void { + websocketRequestCount += 1; + const body: unknown = JSON.parse(data); + if (websocketRequestCount === 1) { + firstRequest = requireRecord(body, "websocket request"); + } + if (websocketRequestCount === 3) { + this.sendJson({ + type: "response.failed", + response: { error: { code: "invalid_request_error", message: "isolated compaction failed" } }, + }); + return; + } + this.emitCodexResponse({ + messageId: `msg_ws_${websocketRequestCount}`, + responseId: `resp_ws_${websocketRequestCount}`, + text: "Hello WS", + }); } } @@ -2248,12 +2341,65 @@ describe("openai-codex streaming", () => { messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], }; const providerSessionState = new Map(); + const midTurnCompaction: CodexCompactionRequestContext = { + operationId: "isolated-success", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "mid_turn", + strategy: "memento", + }; const first = await streamOpenAICodexResponses(websocketModel, context, { fetch: fetchMock as FetchImpl, apiKey: token, sessionId: "ws-handshake-session", providerSessionState, }).result(); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + const isolatedSuccess = await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-handshake-session", + providerSessionState, + codexCompaction: midTurnCompaction, + }).result(); + expect(isolatedSuccess.stopReason).toBe("stop"); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + expect(websocketInstances[1]?.readyState).toBe(MockWebSocket.CLOSED); + expect(websocketInstances[1]?.options?.headers?.["x-codex-turn-state"]).toBe("ws-turn-state-1"); + expect(websocketInstances[1]?.options?.headers?.["x-models-etag"]).toBe("models-etag-1"); + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId: "ws-handshake-session", + compaction: midTurnCompaction, + }); + const isolatedFailure = await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-handshake-session", + providerSessionState, + codexCompaction: { + operationId: "isolated-failure", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "mid_turn", + strategy: "memento", + }, + }).result(); + expect(isolatedFailure.stopReason).toBe("error"); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + expect(websocketInstances[2]?.readyState).toBe(MockWebSocket.CLOSED); + expect(websocketConstructorCount).toBe(3); + expect( + getOpenAICodexTransportDetails(websocketModel, { + sessionId: "ws-handshake-session", + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + hasTurnState: true, + }); // Turn-state is scoped to the current turn, so the SSE replay must be a // within-turn continuation (trailing tool result) to carry the header. const followUp: Context = { @@ -2285,6 +2431,155 @@ describe("openai-codex streaming", () => { providerSessionState, }).result(); expect(fetchMock).toHaveBeenCalledTimes(1); + if (!firstRequest || !continuationRequest || !continuationHeaders) { + throw new Error("expected both Codex transport requests"); + } + const firstMetadata = requireRecord(firstRequest.client_metadata, "first client_metadata"); + const continuationMetadata = requireRecord(continuationRequest.client_metadata, "continuation client_metadata"); + const firstTurnMetadata = parseTurnMetadata(firstMetadata); + const continuationTurnMetadata = parseTurnMetadata(continuationMetadata); + expect(continuationMetadata).toMatchObject({ + "x-codex-installation-id": TEST_INSTALLATION_ID, + session_id: firstMetadata.session_id, + thread_id: firstMetadata.thread_id, + turn_id: firstMetadata.turn_id, + }); + expect(continuationTurnMetadata).toMatchObject({ + installation_id: TEST_INSTALLATION_ID, + session_id: firstTurnMetadata.session_id, + thread_id: firstTurnMetadata.thread_id, + turn_id: firstTurnMetadata.turn_id, + window_id: continuationMetadata["x-codex-window-id"], + request_kind: "turn", + turn_started_at_unix_ms: context.messages[0]?.timestamp, + }); + expect(typeof continuationMetadata["x-codex-window-id"]).toBe("string"); + expect(continuationMetadata["x-codex-window-id"]).not.toBe(firstMetadata["x-codex-window-id"]); + expect(firstMetadata.session_id).toBe(continuationHeaders.get("session-id")); + expect(firstMetadata.thread_id).toBe(continuationHeaders.get("thread-id")); + expect(continuationMetadata["x-codex-window-id"]).toBe(continuationHeaders.get("x-codex-window-id")); + expect(continuationMetadata["x-codex-turn-metadata"]).toBe(continuationHeaders.get("x-codex-turn-metadata")); + }); + + it("clears stale main turn-state after pre-turn compaction", async () => { + const tempDir = TempDir.createSync("@pi-codex-stream-"); + setAgentDir(tempDir.path()); + const token = createCodexTestToken(); + const websocketInstances: MockWebSocket[] = []; + let websocketRequestCount = 0; + + class PreTurnCompactionWebSocket extends MockWebSocket { + handshakeHeaders = { + "x-codex-turn-state": "stale-main-turn-state", + "x-models-etag": "models-etag-1", + }; + + constructor(url: string, options?: { headers?: WsHeaders }) { + super(url, options); + websocketInstances.push(this); + queueMicrotask(() => { + this.readyState = MockWebSocket.OPEN; + this.emit("open", new Event("open")); + }); + } + + send(_data: string): void { + websocketRequestCount += 1; + this.emitCodexResponse({ + messageId: `msg_pre_turn_${websocketRequestCount}`, + responseId: `resp_pre_turn_${websocketRequestCount}`, + text: "Hello WS", + }); + } + } + + global.WebSocket = PreTurnCompactionWebSocket as unknown as typeof WebSocket; + const websocketModel = createCodexTestModel("https://chatgpt.com/backend-api"); + const sseModel: Model<"openai-codex-responses"> = buildModel({ + id: websocketModel.id, + name: websocketModel.name, + api: "openai-codex-responses", + provider: websocketModel.provider, + baseUrl: websocketModel.baseUrl, + reasoning: true, + preferWebsockets: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128000, + maxTokens: 128000, + }); + const providerSessionState = new Map(); + const sessionId = "pre-turn-reset-session"; + let sseHeaders: Headers | undefined; + const fetchMock = vi.fn(async (_input: string | URL, init?: RequestInit) => { + sseHeaders = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers); + return new Response(createCompletedCodexSse("Hello SSE"), { + headers: { "content-type": "text/event-stream" }, + }); + }); + const compaction: CodexCompactionRequestContext = { + operationId: "pre-turn-reset-operation", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }; + + try { + await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock as FetchImpl, + sessionId, + providerSessionState, + }).result(); + await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock as FetchImpl, + sessionId, + providerSessionState, + codexCompaction: compaction, + }).result(); + expect(fetchMock).not.toHaveBeenCalled(); + expect(websocketInstances).toHaveLength(2); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + expect(websocketInstances[1]?.readyState).toBe(MockWebSocket.CLOSED); + expect(websocketInstances[1]?.options?.headers?.["x-codex-turn-state"]).toBeUndefined(); + expect(websocketInstances[1]?.options?.headers?.["x-models-etag"]).toBe("models-etag-1"); + + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId, + compaction, + }); + expect( + getOpenAICodexTransportDetails(websocketModel, { + sessionId, + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + hasTurnState: false, + }); + await streamOpenAICodexResponses( + sseModel, + { + systemPrompt: ["You are a helpful assistant."], + messages: [{ role: "user", content: "Continue after compaction", timestamp: Date.now() }], + }, + { + apiKey: token, + fetch: fetchMock as FetchImpl, + sessionId, + providerSessionState, + }, + ).result(); + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(sseHeaders?.get("x-codex-turn-state")).toBeNull(); + } finally { + for (const state of providerSessionState.values()) state.close(); + providerSessionState.clear(); + } }); it("includes service_tier in websocket payloads when requested", async () => { @@ -2472,6 +2767,17 @@ describe("openai-codex streaming", () => { expect(deltaItems[0]?.role).toBe("user"); expect(JSON.stringify(deltaItems)).toContain("Second question"); expect(JSON.stringify(deltaItems)).not.toContain("First answer"); + const firstMetadata = requireRecord(sentRequests[0]?.client_metadata, "first client_metadata"); + const secondMetadata = requireRecord(sentRequests[1]?.client_metadata, "second client_metadata"); + expect(secondMetadata).toMatchObject({ + "x-codex-installation-id": firstMetadata["x-codex-installation-id"], + session_id: firstMetadata.session_id, + thread_id: firstMetadata.thread_id, + "x-codex-window-id": firstMetadata["x-codex-window-id"], + }); + expect(secondMetadata.turn_id).not.toBe(firstMetadata.turn_id); + expect(parseTurnMetadata(firstMetadata).turn_started_at_unix_ms).toBe(firstContext.messages[0]?.timestamp); + expect(parseTurnMetadata(secondMetadata).turn_started_at_unix_ms).toBe(secondContext.messages.at(-1)?.timestamp); const stats = getOpenAICodexWebSocketDebugStats(model, { sessionId: "ws-delta-session", @@ -3938,10 +4244,12 @@ describe("openai-codex streaming", () => { let constructorCount = 0; let sendCount = 0; + let prewarmHeaders: WsHeaders | undefined; class ReusableWebSocket extends MockWebSocket { constructor(url: string, options?: { headers?: WsHeaders }) { super(url, options); constructorCount += 1; + prewarmHeaders = options?.headers; this.scheduleOpen(); } @@ -3979,6 +4287,11 @@ describe("openai-codex streaming", () => { sessionId: "ws-reuse-session", providerSessionState, }); + expect(prewarmHeaders?.["session-id"]).toBe("ws-reuse-session"); + expect(prewarmHeaders?.["thread-id"]).toBeDefined(); + expect(prewarmHeaders?.["x-codex-window-id"]).toBeDefined(); + expect(prewarmHeaders?.["x-codex-turn-metadata"]).toBeUndefined(); + expect(prewarmHeaders?.["x-codex-installation-id"]).toBeUndefined(); const firstContext: Context = { systemPrompt: ["You are a helpful assistant."], @@ -4016,6 +4329,26 @@ describe("openai-codex streaming", () => { expect(transportDetails.websocketConnected).toBe(true); expect(transportDetails.prewarmed).toBe(true); expect(transportDetails.canAppend).toBe(true); + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId: "ws-reuse-session", + compaction: { + operationId: "history-rewrite", + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + strategy: "memento", + }, + }); + expect( + getOpenAICodexTransportDetails(model, { + sessionId: "ws-reuse-session", + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + canAppend: false, + }); }); it("scopes x-codex-turn-state to the current turn on SSE requests", async () => { diff --git a/packages/ai/test/openai-responses-history-payload.test.ts b/packages/ai/test/openai-responses-history-payload.test.ts index afdedd481..9520fabf0 100644 --- a/packages/ai/test/openai-responses-history-payload.test.ts +++ b/packages/ai/test/openai-responses-history-payload.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { convertCodexResponsesMessages, streamOpenAICodexResponses, @@ -9,6 +9,17 @@ import type { Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/ import { createOpenAIResponsesHistoryPayload, truncateResponseItemId } from "@oh-my-pi/pi-ai/utils"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import * as piUtils from "@oh-my-pi/pi-utils"; + +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); function createAbortedSignal(): AbortSignal { const controller = new AbortController(); diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index 63bc5f331..757a7f31f 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -288,7 +288,13 @@ describe("processResponsesStream: lost output_item.added recovery", () => { { type: "response.output_item.done", output_index: 0, - item: { type: "reasoning", summary: [{ type: "summary_text", text: "first" }] }, + item: { + type: "reasoning", + summary: [ + { type: "summary_text", text: "Plan" }, + { type: "summary_text", text: "Planning details" }, + ], + }, }, { type: "response.output_item.done", @@ -305,7 +311,7 @@ describe("processResponsesStream: lost output_item.added recovery", () => { expect(output.content).toHaveLength(2); const [first, second] = output.content; if (first?.type !== "thinking" || second?.type !== "thinking") throw new Error("expected thinking blocks"); - expect(first.thinking).toBe("first"); + expect(first.thinking).toBe("Plan\n\nPlanning details"); expect(second.thinking).toBe("second"); expect(first.thinkingSignature).toBeDefined(); expect(second.thinkingSignature).toBeDefined(); diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 706e3b509..735714ef3 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -31524,7 +31524,7 @@ }, "openai/gpt-5.6-sol": { "id": "openai/gpt-5.6-sol", - "name": "GPT-5.6 Sol", + "name": "GPT-5.6 Sol (new)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -73972,13 +73972,13 @@ "text" ], "cost": { - "input": 0.9199999999999999, - "output": 3, - "cacheRead": 0.18, + "input": 0.84, + "output": 2.64, + "cacheRead": 0.156, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 1048576, + "maxTokens": 128000, "thinking": { "mode": "effort", "efforts": [ diff --git a/packages/catalog/src/wire/codex.ts b/packages/catalog/src/wire/codex.ts index c2336b6d2..329ace70f 100644 --- a/packages/catalog/src/wire/codex.ts +++ b/packages/catalog/src/wire/codex.ts @@ -10,6 +10,13 @@ export const OPENAI_HEADERS = { ORIGINATOR: "originator", SESSION_ID: "session_id", CONVERSATION_ID: "conversation_id", + SCOPED_SESSION_ID: "session-id", + THREAD_ID: "thread-id", + INSTALLATION_ID: "x-codex-installation-id", + WINDOW_ID: "x-codex-window-id", + TURN_METADATA: "x-codex-turn-metadata", + PARENT_THREAD_ID: "x-codex-parent-thread-id", + SUBAGENT: "x-openai-subagent", /** Responses Lite transport marker (codex-rs `add_responses_lite_header`); value is always `"true"`. */ RESPONSES_LITE: "x-openai-internal-codex-responses-lite", } as const; diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..edac6e685 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -84,6 +84,7 @@ import type { AssistantMessageEvent, AssistantRetryRecovery, AssistantRetryRecoveryKind, + CodexCompactionContext, Context, ImageContent, Message, @@ -117,6 +118,7 @@ import { streamSimple, } from "@oh-my-pi/pi-ai"; import * as AIError from "@oh-my-pi/pi-ai/error"; +import { resetOpenAICodexHistoryAfterCompaction } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema"; import { GeminiHeaderRunDetector, isGeminiThinkingModel } from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { type RepeatedToolCallDetection, ToolCallLoopGuard } from "@oh-my-pi/pi-ai/utils/tool-call-loop-guard"; @@ -602,6 +604,20 @@ function compactionDeadEndWarning(remedies: string): string { ); } +function createCodexCompactionContext(options: { + trigger: CodexCompactionContext["trigger"]; + reason: CodexCompactionContext["reason"]; + phase: CodexCompactionContext["phase"]; +}): CodexCompactionContext { + return { + operationId: crypto.randomUUID(), + trigger: options.trigger, + reason: options.reason, + phase: options.phase, + strategy: "memento", + }; +} + /** * Per-turn prune cache window. A tool result whose all-message suffix exceeds * this is in the warm, already-sent prompt-cache prefix: re-writing it costs the @@ -2871,6 +2887,12 @@ export class AgentSession { // compaction path so the advisor model's maintenance call also emits spans. const telemetry = resolveTelemetry(agent.telemetry, advisorSessionId); + const codexCompaction = createCodexCompactionContext({ + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + }); + for (const candidate of candidates) { const apiKey = await this.#modelRegistry.getApiKey(candidate, advisorSessionId); if (!apiKey) continue; @@ -2889,6 +2911,8 @@ export class AgentSession { tools: agent.state.tools, sessionId: advisorSessionId, promptCacheKey: advisorSessionId, + providerSessionState: this.#providerSessionState, + codexCompaction, }, ); break; @@ -9751,6 +9775,7 @@ export class AgentSession { let firstKeptEntryId: string; let tokensBefore: number; let details: unknown; + let codexCompaction: CodexCompactionContext | undefined; // Snapcompact runs locally first. The frame cap is sized from the live // model window via #computeSnapcompactMaxFrames so the post-render context @@ -9830,6 +9855,11 @@ export class AgentSession { details = snapcompactResult.details; preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) }; } else { + codexCompaction = createCodexCompactionContext({ + trigger: "manual", + reason: "user_requested", + phase: "standalone_turn", + }); // Generate compaction result. Only convert known abort-shaped // rejections (AbortError raised while the abort signal is set, // or an already-typed sentinel) into `CompactionCancelledError` @@ -9851,6 +9881,7 @@ export class AgentSession { extraContext: compactionPrep.hookContext, remoteInstructions: this.#baseSystemPrompt.join("\n\n"), convertToLlm: messages => this.#convertToLlmForSideRequest(messages), + codexCompaction, }, compactionCandidates, ); @@ -9893,7 +9924,11 @@ export class AgentSession { this.#planReferenceSent = false; this.#resetAllAdvisorRuntimes(); this.#syncTodoPhasesFromBranch(); - this.#closeCodexProviderSessionsForHistoryRewrite(); + if (codexCompaction) { + this.#resetCodexProviderAfterCompaction(codexCompaction); + } else { + this.#closeCodexProviderSessionsForHistoryRewrite(); + } // Get the saved compaction entry for the hook const savedCompactionEntry = newEntries.find(e => e.type === "compaction" && e.summary === summary) as @@ -10259,6 +10294,7 @@ export class AgentSession { await this.#runAutoCompaction("threshold", false, false, false, { autoContinue: false, triggerContextTokens: contextTokens, + phase: "pre_turn", }); } @@ -10333,6 +10369,7 @@ export class AgentSession { suppressContinuation: true, suppressHandoff: true, triggerContextTokens: contextTokens, + phase: "mid_turn", }); if (signal?.aborted) return; @@ -10579,6 +10616,7 @@ export class AgentSession { return await this.#runAutoCompaction("threshold", false, false, allowDefer, { autoContinue, triggerContextTokens: postMaintenanceContextTokens, + phase: "pre_turn", }); } logger.debug("Auto-compaction threshold satisfied but context promotion took over", { @@ -10939,7 +10977,11 @@ export class AgentSession { ): Promise { const compactionEntryBefore = getLatestCompactionEntry(this.sessionManager.getBranch()); await this.#dropPersistedAssistantTurn(assistantMessage); - const result = await this.#runAutoCompaction(reason, true, false, allowDefer, options); + const result = await this.#runAutoCompaction(reason, true, false, allowDefer, { + autoContinue: options.autoContinue, + triggerContextTokens: options.triggerContextTokens, + phase: "mid_turn", + }); const compactionEntryAfter = getLatestCompactionEntry(this.sessionManager.getBranch()); if (result.historyRewritten !== true && compactionEntryAfter === compactionEntryBefore) { this.#restoreFailedAssistantTurn(assistantMessage); @@ -11551,6 +11593,14 @@ export class AgentSession { this.#closeProviderSessionsForModelSwitch(currentModel, currentModel); } + #resetCodexProviderAfterCompaction(compaction: CodexCompactionContext): void { + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState: this.#providerSessionState, + sessionId: this.sessionId, + compaction, + }); + } + #resetCurrentResponsesProviderSession(reason: string): void { const currentModel = this.model; if (currentModel?.api !== "openai-responses" && currentModel?.api !== "openai-codex-responses") { @@ -11982,6 +12032,7 @@ export class AgentSession { tools: this.agent.state.tools, sessionId: this.sessionId, promptCacheKey: this.sessionId, + providerSessionState: this.#providerSessionState, // Route every summarization HTTP request through the // session's side-stream transport so the provider // concurrency cap (e.g. providers.ollama-cloud.maxConcurrency) @@ -12320,6 +12371,7 @@ export class AgentSession { triggerContextTokens?: number; suppressContinuation?: boolean; suppressHandoff?: boolean; + phase?: CodexCompactionContext["phase"]; } = {}, ): Promise { const compactionSettings = this.settings.getGroup("compaction"); @@ -12362,7 +12414,7 @@ export class AgentSession { async signal => { await Promise.resolve(); if (signal.aborted) return; - await this.#runAutoCompaction(reason, willRetry, true); + await this.#runAutoCompaction(reason, willRetry, true, true, { phase: options.phase }); }, { generation }, ); @@ -12509,6 +12561,7 @@ export class AgentSession { let hookCompaction: CompactionResult | undefined; let fromExtension = false; let preserveData: Record | undefined; + let codexCompaction: CodexCompactionContext | undefined; if (this.#extensionRunner?.hasHandlers("session_before_compact")) { const hookResult = (await this.#extensionRunner.emit({ @@ -12647,6 +12700,13 @@ export class AgentSession { const telemetry = resolveTelemetry(this.agent.telemetry, this.sessionId); let compactResult: CompactionResult | undefined; let lastError: unknown; + codexCompaction = createCodexCompactionContext({ + trigger: "auto", + reason: "context_limit", + phase: + options.phase ?? + (reason === "threshold" ? "pre_turn" : reason === "idle" ? "standalone_turn" : "mid_turn"), + }); for (let candidateIndex = 0; candidateIndex < candidates.length; candidateIndex++) { const candidate = candidates[candidateIndex]; @@ -12679,6 +12739,8 @@ export class AgentSession { tools: this.agent.state.tools, sessionId: this.sessionId, promptCacheKey: this.sessionId, + providerSessionState: this.#providerSessionState, + codexCompaction, }, ); break; @@ -12797,7 +12859,11 @@ export class AgentSession { this.#planReferenceSent = false; this.#resetAllAdvisorRuntimes(); this.#syncTodoPhasesFromBranch(); - this.#closeCodexProviderSessionsForHistoryRewrite(); + if (codexCompaction) { + this.#resetCodexProviderAfterCompaction(codexCompaction); + } else { + this.#closeCodexProviderSessionsForHistoryRewrite(); + } // Get the saved compaction entry for the hook const savedCompactionEntry = newEntries.find(e => e.type === "compaction" && e.summary === summary) as diff --git a/packages/coding-agent/test/agent-session-eager-compaction.test.ts b/packages/coding-agent/test/agent-session-eager-compaction.test.ts index a4d0c2db2..cd8528ac9 100644 --- a/packages/coding-agent/test/agent-session-eager-compaction.test.ts +++ b/packages/coding-agent/test/agent-session-eager-compaction.test.ts @@ -2,7 +2,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core"; import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction"; -import type { TextContent } from "@oh-my-pi/pi-ai"; +import type { Model, TextContent } from "@oh-my-pi/pi-ai"; +import * as codexResponses from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -135,22 +136,23 @@ describe("AgentSession eager prelude re-injection after compaction", () => { async function createHarness( settingsOverride: Record = {}, - opts: { agentId?: string; agentKind?: "main" | "sub" } = {}, + opts: { agentId?: string; agentKind?: "main" | "sub"; model?: Model } = {}, ): Promise { const observedCalls: ObservedPromptCall[] = []; const waiters: Array<{ predicate: (call: ObservedPromptCall) => boolean; resolve: (call: ObservedPromptCall) => void; }> = []; - const bundled = getBundledModel("anthropic", "claude-sonnet-4-5"); - if (!bundled) throw new Error("Expected claude-sonnet-4-5 model to exist"); + const defaultModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!defaultModel) throw new Error("Expected claude-sonnet-4-5 model to exist"); + const selectedModel = opts.model ?? defaultModel; // Pin the window and output reservation: usage figures below trip the // context-full strategy at a 200k/64k threshold; catalog regeneration must // not shift the headroom math. - const model = { ...bundled, contextWindow: 200_000, maxTokens: 64_000 }; + const model = { ...selectedModel, contextWindow: 200_000, maxTokens: 64_000 }; const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${cleanups.length}.db`)); - authStorage.setRuntimeApiKey("anthropic", "test-key"); + authStorage.setRuntimeApiKey(model.provider, "test-key"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), `models-${cleanups.length}.yml`)); const settings = Settings.isolated({ "compaction.enabled": true, @@ -359,4 +361,25 @@ describe("AgentSession eager prelude re-injection after compaction", () => { expect(continuation.messageTexts.some(text => text.includes("Consider calling"))).toBe(false); expect(continuation.messageTexts.some(text => text.includes("You MUST call"))).toBe(false); }); + + it("resets Codex provider history after successful auto-compaction", async () => { + const model = getBundledModel("openai-codex", "gpt-5.6-terra"); + if (!model) throw new Error("Expected gpt-5.6-terra model to exist"); + const resetSpy = vi.spyOn(codexResponses, "resetOpenAICodexHistoryAfterCompaction"); + const { session, waitForCall } = await createHarness({}, { model }); + stubCompaction(); + + await runToContinuation(session, waitForCall); + + expect(resetSpy).toHaveBeenCalledTimes(1); + const reset = resetSpy.mock.calls[0]?.[0]; + if (!reset) throw new Error("Expected Codex compaction reset"); + expect(reset.providerSessionState).toBe(session.providerSessionState); + expect(reset.sessionId).toBe(session.sessionId); + expect(reset.compaction).toMatchObject({ + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + }); + }); }); From f79098b9ba1aa1dde9cc622f3a8aa5d8e7f25e72 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 11:56:38 +0200 Subject: [PATCH 052/205] fix(providers): corrected novita discovery and login - Corrected Novita pricing from ten-thousandths of a dollar per million tokens. - Validated pasted keys against the authenticated balance endpoint. - Made live discovery authoritative and excluded models without positive output limits. --- packages/ai/src/registry/novita.ts | 9 +- packages/ai/test/novita-login.test.ts | 74 + packages/catalog/src/models.json | 5706 ++++++++--------- .../src/provider-models/descriptors.ts | 1 + .../src/provider-models/openai-compat.ts | 12 +- packages/catalog/test/novita-provider.test.ts | 43 +- 6 files changed, 2950 insertions(+), 2895 deletions(-) create mode 100644 packages/ai/test/novita-login.test.ts diff --git a/packages/ai/src/registry/novita.ts b/packages/ai/src/registry/novita.ts index 3b573dab6..a4c6abf23 100644 --- a/packages/ai/src/registry/novita.ts +++ b/packages/ai/src/registry/novita.ts @@ -9,13 +9,14 @@ export const loginNovita = createApiKeyLogin({ placeholder: "sk_...", validation: { kind: "models-endpoint", - provider: "novita", - modelsUrl: "https://api.novita.ai/openai/v1/models", + provider: "Novita", + modelsUrl: "https://api.novita.ai/openapi/v1/billing/balance/detail", + headers: { "Content-Type": "application/json" }, }, }); export const novitaProvider = { id: "novita", name: "Novita", - login: (cb: Parameters[0]) => loginNovita(cb), -} as const satisfies ProviderDefinition; + login: loginNovita, +} satisfies ProviderDefinition & { readonly id: "novita" }; diff --git a/packages/ai/test/novita-login.test.ts b/packages/ai/test/novita-login.test.ts new file mode 100644 index 000000000..810b27870 --- /dev/null +++ b/packages/ai/test/novita-login.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, test, vi } from "bun:test"; +import { loginNovita } from "../src/registry/novita"; +import { getOAuthProviders } from "../src/registry/oauth"; +import type { FetchImpl } from "../src/types"; + +describe("Novita login", () => { + test("registers Novita as an available API-key provider", () => { + const provider = getOAuthProviders().find(item => item.id === "novita"); + expect(provider).toMatchObject({ id: "novita", name: "Novita", available: true }); + }); + + test("validates the pasted key against the authenticated balance endpoint", async () => { + const authEvents: Array<{ url: string; instructions?: string }> = []; + const prompts: Array<{ message: string; placeholder?: string }> = []; + const progress: string[] = []; + const requests: Array<{ + url: string; + method: string | undefined; + authorization: string | null; + contentType: string | null; + }> = []; + const fetchMock: FetchImpl = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const headers = new Headers(init?.headers); + requests.push({ + url: String(input), + method: init?.method, + authorization: headers.get("authorization"), + contentType: headers.get("content-type"), + }); + return Response.json({ availableBalance: "0" }); + }); + + const apiKey = await loginNovita({ + onAuth: info => authEvents.push(info), + onPrompt: async prompt => { + prompts.push(prompt); + return " novita-test-key "; + }, + onProgress: message => progress.push(message), + fetch: fetchMock, + }); + + expect(apiKey).toBe("novita-test-key"); + expect(authEvents).toEqual([ + { + url: "https://novita.ai/settings/key-management", + instructions: "Create or copy your API key from the Novita dashboard", + }, + ]); + expect(prompts).toEqual([{ message: "Paste your Novita API key", placeholder: "sk_..." }]); + expect(progress).toEqual(["Validating API key..."]); + expect(requests).toEqual([ + { + url: "https://api.novita.ai/openapi/v1/billing/balance/detail", + method: "GET", + authorization: "Bearer novita-test-key", + contentType: "application/json", + }, + ]); + }); + + test("rejects a key rejected by Novita", async () => { + const fetchMock: FetchImpl = vi.fn(async () => + Response.json({ code: 401, reason: "UNAUTHORIZED", message: "key not found", metadata: {} }, { status: 401 }), + ); + + await expect( + loginNovita({ + onPrompt: async () => "invalid-novita-key", + fetch: fetchMock, + }), + ).rejects.toThrow("Novita API key validation failed (401)"); + }); +}); diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 15a20cea1..972fa5e15 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -54811,6 +54811,2849 @@ } } }, + "novita": { + "baichuan/baichuan-m2-32b": { + "id": "baichuan/baichuan-m2-32b", + "name": "BaiChuan M2 32B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.07, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072, + "supportsTools": false + }, + "baidu/cobuddy": { + "id": "baidu/cobuddy", + "name": "CoBuddy", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.28, + "output": 1.13, + "cacheRead": 0.07, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "baidu/ernie-4.5-21B-a3b": { + "id": "baidu/ernie-4.5-21B-a3b", + "name": "ERNIE 4.5 21B A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.28, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 120000, + "maxTokens": 8000, + "supportsTools": true + }, + "baidu/ernie-4.5-vl-424b-a47b": { + "id": "baidu/ernie-4.5-vl-424b-a47b", + "name": "ERNIE 4.5 VL 424B A47B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.42, + "output": 1.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 123000, + "maxTokens": 16000, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "bunny": { + "id": "bunny", + "name": "Bunny", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "deepseek/deepseek_v3": { + "id": "deepseek/deepseek_v3", + "name": "DeepSeek V3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.89, + "output": 0.89, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true + }, + "deepseek/deepseek-ocr": { + "id": "deepseek/deepseek-ocr", + "name": "DeepSeek-OCR", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.03, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "deepseek/deepseek-ocr-2": { + "id": "deepseek/deepseek-ocr-2", + "name": "DeepSeek-OCR 2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.03, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "deepseek/deepseek-r1": { + "id": "deepseek/deepseek-r1", + "name": "R1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 4, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-0528": { + "id": "deepseek/deepseek-r1-0528", + "name": "R1 0528", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 2.5, + "cacheRead": 0.35, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-0528-qwen3-8b": { + "id": "deepseek/deepseek-r1-0528-qwen3-8b", + "name": "DeepSeek R1 0528 Qwen3 8B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.06, + "output": 0.09, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32000, + "supportsTools": false + }, + "deepseek/deepseek-r1-distill-llama-70b": { + "id": "deepseek/deepseek-r1-distill-llama-70b", + "name": "DeepSeek R1 Distill LLama 70B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.8, + "output": 0.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-turbo": { + "id": "deepseek/deepseek-r1-turbo", + "name": "DeepSeek R1 (Turbo)", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1/community": { + "id": "deepseek/deepseek-r1/community", + "name": "DeepSeek R1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 4, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 8000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3-0324": { + "id": "deepseek/deepseek-v3-0324", + "name": "DeepSeek V3 0324", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 1.12, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true + }, + "deepseek/deepseek-v3-turbo": { + "id": "deepseek/deepseek-v3-turbo", + "name": "DeepSeek V3 (Turbo)", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.4, + "output": 1.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true + }, + "deepseek/deepseek-v3.1": { + "id": "deepseek/deepseek-v3.1", + "name": "DeepSeek V3.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 1, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.1-terminus": { + "id": "deepseek/deepseek-v3.1-terminus", + "name": "DeepSeek V3.1 Terminus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 1, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.2": { + "id": "deepseek/deepseek-v3.2", + "name": "DeepSeek V3.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.269, + "output": 0.4, + "cacheRead": 0.1345, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.2-exp": { + "id": "deepseek/deepseek-v3.2-exp", + "name": "DeepSeek V3.2 Exp", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 0.41, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3/community": { + "id": "deepseek/deepseek-v3/community", + "name": "DeepSeek V3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.89, + "output": 0.89, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 8000, + "supportsTools": true + }, + "deepseek/deepseek-v4-flash": { + "id": "deepseek/deepseek-v4-flash", + "name": "DeepSeek V4 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0.028, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v4-pro": { + "id": "deepseek/deepseek-v4-pro", + "name": "DeepSeek V4 Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.6, + "output": 3.2, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "dev/glm46": { + "id": "dev/glm46", + "name": "dev/glm46", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 256000, + "supportsTools": true + }, + "google/gemma-3-12b-it": { + "id": "google/gemma-3-12b-it", + "name": "Gemma3 12B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.05, + "output": 0.1, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 8192, + "supportsTools": false + }, + "google/gemma-3-27b-it": { + "id": "google/gemma-3-27b-it", + "name": "Gemma 3 27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.119, + "output": 0.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 98304, + "maxTokens": 16384, + "supportsTools": false + }, + "google/gemma-4-26b-a4b-it": { + "id": "google/gemma-4-26b-a4b-it", + "name": "Gemma 4 26B A4B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.13, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "google/gemma-4-31b-it": { + "id": "google/gemma-4-31b-it", + "name": "Gemma 4 31B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.14, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gryphe/mythomax-l2-13b": { + "id": "gryphe/mythomax-l2-13b", + "name": "Mythomax L2 13B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.09, + "output": 0.09, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 4096, + "maxTokens": 3200, + "supportsTools": false + }, + "gt-4p": { + "id": "gt-4p", + "name": "gt-4p", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": 131072, + "supportsTools": true + }, + "inclusionai/ling-2.6-1t": { + "id": "inclusionai/ling-2.6-1t", + "name": "Ling-2.6-1T", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "inclusionai/ling-2.6-flash": { + "id": "inclusionai/ling-2.6-flash", + "name": "Ling-2.6 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.1, + "output": 0.3, + "cacheRead": 0.02, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "inclusionai/ring-2.6-1t": { + "id": "inclusionai/ring-2.6-1t", + "name": "Ring-2.6-1T", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "kwaipilot/kat-coder-pro": { + "id": "kwaipilot/kat-coder-pro", + "name": "Kat Coder Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 128000, + "supportsTools": true + }, + "meta-llama/llama-3.1-8b-instruct": { + "id": "meta-llama/llama-3.1-8b-instruct", + "name": "Llama 3.1 8B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.02, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 16384, + "supportsTools": false + }, + "meta-llama/llama-3.2-1b-instruct": { + "id": "meta-llama/llama-3.2-1b-instruct", + "name": "Llama 3.2 1B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.02, + "output": 0.02, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131000, + "maxTokens": 32000, + "supportsTools": false + }, + "meta-llama/llama-3.2-3b-instruct": { + "id": "meta-llama/llama-3.2-3b-instruct", + "name": "Llama 3.2 3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.03, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32768, + "maxTokens": 32000, + "supportsTools": false + }, + "meta-llama/llama-3.3-70b-instruct": { + "id": "meta-llama/llama-3.3-70b-instruct", + "name": "Llama 3.3 70B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.135, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 6000, + "maxTokens": 120000, + "supportsTools": true + }, + "meta-llama/llama-4-maverick-17b-128e-instruct-fp8": { + "id": "meta-llama/llama-4-maverick-17b-128e-instruct-fp8", + "name": "Llama 4 Maverick Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.27, + "output": 0.85, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 8192, + "supportsTools": false + }, + "meta-llama/llama-4-scout-17b-16e-instruct": { + "id": "meta-llama/llama-4-scout-17b-16e-instruct", + "name": "Llama 4 Scout 17B 16E", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.18, + "output": 0.59, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072, + "supportsTools": false + }, + "microsoft/wizardlm-2-8x22b": { + "id": "microsoft/wizardlm-2-8x22b", + "name": "Wizardlm 2 8x22B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.62, + "output": 0.62, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65535, + "maxTokens": 8000, + "supportsTools": false + }, + "minimax/minimax-m2": { + "id": "minimax/minimax-m2", + "name": "MiniMax M2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.1": { + "id": "minimax/minimax-m2.1", + "name": "MiniMax M2.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.5": { + "id": "minimax/minimax-m2.5", + "name": "MiniMax M2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131100, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.5-highspeed": { + "id": "minimax/minimax-m2.5-highspeed", + "name": "MiniMax M2.5-highspeed", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131100, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.7": { + "id": "minimax/minimax-m2.7", + "name": "MiniMax M2.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.7-highspeed": { + "id": "minimax/minimax-m2.7-highspeed", + "name": "MiniMax M2.7 highspeed", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax M3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "minimaxai/minimax-m1-80k": { + "id": "minimaxai/minimax-m1-80k", + "name": "MiniMax M1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 2.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 40000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "mistralai/mistral-nemo": { + "id": "mistralai/mistral-nemo", + "name": "Mistral Nemo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.04, + "output": 0.17, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 60288, + "maxTokens": 16000, + "supportsTools": false + }, + "moonshotai/kimi-k2-0905": { + "id": "moonshotai/kimi-k2-0905", + "name": "Kimi K2 0905", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 100352, + "supportsTools": true + }, + "moonshotai/kimi-k2-instruct": { + "id": "moonshotai/kimi-k2-instruct", + "name": "Kimi K2 Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.57, + "output": 2.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 100352, + "supportsTools": true + }, + "moonshotai/kimi-k2-thinking": { + "id": "moonshotai/kimi-k2-thinking", + "name": "Kimi K2 Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 100352, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "moonshotai/kimi-k2.5": { + "id": "moonshotai/kimi-k2.5", + "name": "Kimi K2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/kimi-k2.6": { + "id": "moonshotai/kimi-k2.6", + "name": "Kimi K2.6", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.8, + "output": 3.4, + "cacheRead": 0.16, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/kimi-k2.7-code": { + "id": "moonshotai/kimi-k2.7-code", + "name": "Kimi K2.7 Code", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.95, + "output": 4, + "cacheRead": 0.19, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "nousresearch/hermes-2-pro-llama-3-8b": { + "id": "nousresearch/hermes-2-pro-llama-3-8b", + "name": "Hermes 2 Pro Llama 3 8B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.14, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "nvidia/nemotron-3-nano-30b-a3b": { + "id": "nvidia/nemotron-3-nano-30b-a3b", + "name": "Nemotron 3 Nano 30B A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-oss-120b": { + "id": "openai/gpt-oss-120b", + "name": "GPT OSS 120B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "openai/gpt-oss-20b": { + "id": "openai/gpt-oss-20b", + "name": "GPT OSS 20B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.04, + "output": 0.15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "paddlepaddle/paddleocr-vl": { + "id": "paddlepaddle/paddleocr-vl", + "name": "PaddleOCR-VL", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.02, + "output": 0.02, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 16384, + "supportsTools": false + }, + "qwen/qwen-2.5-72b-instruct": { + "id": "qwen/qwen-2.5-72b-instruct", + "name": "Qwen2.5 72B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.38, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": 8192, + "supportsTools": true + }, + "qwen/qwen-mt-plus": { + "id": "qwen/qwen-mt-plus", + "name": "Qwen MT Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.25, + "output": 0.75, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 8192, + "supportsTools": false + }, + "qwen/qwen3-235b-a22b-fp8": { + "id": "qwen/qwen3-235b-a22b-fp8", + "name": "Qwen3 235B A22B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 0.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 40960, + "maxTokens": 20000, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3-235b-a22b-instruct-2507": { + "id": "qwen/qwen3-235b-a22b-instruct-2507", + "name": "Qwen3 235B A22B Instruct 2507", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.09, + "output": 0.58, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 16384, + "supportsTools": true + }, + "qwen/qwen3-235b-a22b-thinking-2507": { + "id": "qwen/qwen3-235b-a22b-thinking-2507", + "name": "Qwen3 235B A22B Thinking 2507", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-coder-30b-a3b-instruct": { + "id": "qwen/qwen3-coder-30b-a3b-instruct", + "name": "Qwen3 Coder 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.27, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 160000, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-coder-480b-a35b-instruct": { + "id": "qwen/qwen3-coder-480b-a35b-instruct", + "name": "Qwen3 Coder 480B A35B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.38, + "output": 1.55, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-coder-next": { + "id": "qwen/qwen3-coder-next", + "name": "Qwen3 Coder Next", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-max": { + "id": "qwen/qwen3-max", + "name": "Qwen3 Max", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 2.11, + "output": 8.45, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-next-80b-a3b-instruct": { + "id": "qwen/qwen3-next-80b-a3b-instruct", + "name": "Qwen3-Next-80B-A3B-Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-omni-30b-a3b-instruct": { + "id": "qwen/qwen3-omni-30b-a3b-instruct", + "name": "Qwen3 Omni 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 0.97, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true + }, + "qwen/qwen3-omni-30b-a3b-thinking": { + "id": "qwen/qwen3-omni-30b-a3b-thinking", + "name": "Qwen3 Omni 30B A3B Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 0.97, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-vl-235b-a22b-instruct": { + "id": "qwen/qwen3-vl-235b-a22b-instruct", + "name": "Qwen3 VL 235B A22B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-vl-235b-a22b-thinking": { + "id": "qwen/qwen3-vl-235b-a22b-thinking", + "name": "Qwen3 VL 235B A22B Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.98, + "output": 3.95, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-vl-30b-a3b-instruct": { + "id": "qwen/qwen3-vl-30b-a3b-instruct", + "name": "Qwen3 VL 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.2, + "output": 0.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3.5-122b-a10b": { + "id": "qwen/qwen3.5-122b-a10b", + "name": "Qwen3.5 122B-A10B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.4, + "output": 3.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-27b": { + "id": "qwen/qwen3.5-27b", + "name": "Qwen3.5-27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-35b-a3b": { + "id": "qwen/qwen3.5-35b-a3b", + "name": "Qwen3.5-35B-A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-397b-a17b": { + "id": "qwen/qwen3.5-397b-a17b", + "name": "Qwen3.5 397B A17B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-plus": { + "id": "qwen/qwen3.5-plus", + "name": "Qwen3.5 Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-27b": { + "id": "qwen/qwen3.6-27b", + "name": "Qwen3.6 27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-35b-a3b": { + "id": "qwen/qwen3.6-35b-a3b", + "name": "Qwen3.6-35B-A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.248, + "output": 1.485, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-plus": { + "id": "qwen/qwen3.6-plus", + "name": "Qwen3.6 Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.7-max": { + "id": "qwen/qwen3.7-max", + "name": "Qwen3.7 Max", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.25, + "output": 3.75, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "sao10k/l3-70b-euryale-v2.1": { + "id": "sao10k/l3-70b-euryale-v2.1", + "name": "L3 70B Euryale V2.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1.48, + "output": 1.48, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": true + }, + "sao10k/l3-8b-lunaris": { + "id": "sao10k/l3-8b-lunaris", + "name": "Sao10k L3 8B Lunaris", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "Sao10K/L3-8B-Stheno-v3.2": { + "id": "Sao10K/L3-8B-Stheno-v3.2", + "name": "L3 8B Stheno V3.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 32000, + "supportsTools": true + }, + "sao10k/l31-70b-euryale-v2.2": { + "id": "sao10k/l31-70b-euryale-v2.2", + "name": "L31 70B Euryale V2.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.48, + "output": 1.48, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "stepfun/step-3.7-flash": { + "id": "stepfun/step-3.7-flash", + "name": "Step 3.7 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.2, + "output": 1.15, + "cacheRead": 0.04, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 256000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "thudm/glm-4-32b-0414": { + "id": "thudm/glm-4-32b-0414", + "name": "GLM-4-32B-0414", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 1.66, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": 32000, + "supportsTools": true + }, + "xiaomimimo/mimo-v2.5": { + "id": "xiaomimimo/mimo-v2.5", + "name": "XiaomiMiMo/MiMo-V2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.168, + "output": 0.336, + "cacheRead": 0.0034, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "xiaomimimo/mimo-v2.5-pro": { + "id": "xiaomimimo/mimo-v2.5-pro", + "name": "XiaomiMiMo/MiMo-V2.5-Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.522, + "output": 1.044, + "cacheRead": 0.0043, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "zai-org/autoglm-phone-9b-multilingual": { + "id": "zai-org/autoglm-phone-9b-multilingual", + "name": "AutoGLM-Phone-9B-Multilingual", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.035, + "output": 0.138, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 65536, + "supportsTools": false + }, + "zai-org/glm-4.5-air": { + "id": "zai-org/glm-4.5-air", + "name": "zai-org/glm-4.5-air", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.13, + "output": 0.85, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 98304, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.5v": { + "id": "zai-org/glm-4.5v", + "name": "GLM 4.5V", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 1.8, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.6": { + "id": "zai-org/glm-4.6", + "name": "GLM 4.6", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 2.2, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.6v": { + "id": "zai-org/glm-4.6v", + "name": "GLM 4.6V", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 0.9, + "cacheRead": 0.055, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7": { + "id": "zai-org/glm-4.7", + "name": "GLM 4.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7-flash": { + "id": "zai-org/glm-4.7-flash", + "name": "GLM 4.7 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.4, + "cacheRead": 0.01, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 128000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7-h": { + "id": "zai-org/glm-4.7-h", + "name": "GLM-4.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5": { + "id": "zai-org/glm-5", + "name": "GLM 5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1, + "output": 3.2, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5-turbo": { + "id": "zai-org/glm-5-turbo", + "name": "GLM-5-Turbo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.2, + "output": 4, + "cacheRead": 0.24, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5.1": { + "id": "zai-org/glm-5.1", + "name": "GLM 5.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.38, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5.2": { + "id": "zai-org/glm-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "xhigh": "max" + } + } + }, + "zai-org/glm-5v-turbo": { + "id": "zai-org/glm-5v-turbo", + "name": "GLM-5V-Turbo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.2, + "output": 4, + "cacheRead": 0.24, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + } + }, "nvidia": { "01-ai/yi-large": { "id": "01-ai/yi-large", @@ -58324,2869 +61167,6 @@ "maxTokens": null } }, - "novita": { - "baichuan/baichuan-m2-32b": { - "id": "baichuan/baichuan-m2-32b", - "name": "BaiChuan M2 32B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.7, - "output": 0.7, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 131072, - "supportsTools": false - }, - "baidu/cobuddy": { - "id": "baidu/cobuddy", - "name": "CoBuddy", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 2.8, - "output": 11.3, - "cacheRead": 0.7, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "baidu/ernie-4.5-21B-a3b": { - "id": "baidu/ernie-4.5-21B-a3b", - "name": "ERNIE 4.5 21B A3B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.7, - "output": 2.8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 120000, - "maxTokens": 8000, - "supportsTools": true - }, - "baidu/ernie-4.5-vl-424b-a47b": { - "id": "baidu/ernie-4.5-vl-424b-a47b", - "name": "ERNIE 4.5 VL 424B A47B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 4.2, - "output": 12.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 123000, - "maxTokens": 16000, - "supportsTools": false, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "bunny": { - "id": "bunny", - "name": "Bunny", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 32768, - "supportsTools": true - }, - "deepseek/deepseek_v3": { - "id": "deepseek/deepseek_v3", - "name": "DeepSeek V3", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 8.9, - "output": 8.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 64000, - "maxTokens": 16000, - "supportsTools": true - }, - "deepseek/deepseek-ocr": { - "id": "deepseek/deepseek-ocr", - "name": "DeepSeek-OCR", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.3, - "output": 0.3, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": false - }, - "deepseek/deepseek-ocr-2": { - "id": "deepseek/deepseek-ocr-2", - "name": "DeepSeek-OCR 2", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.3, - "output": 0.3, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": false - }, - "deepseek/deepseek-r1": { - "id": "deepseek/deepseek-r1", - "name": "R1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 40, - "output": 40, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 64000, - "maxTokens": 16000, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-r1-0528": { - "id": "deepseek/deepseek-r1-0528", - "name": "R1 0528", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 7, - "output": 25, - "cacheRead": 3.5, - "cacheWrite": 0 - }, - "contextWindow": 163840, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-r1-0528-qwen3-8b": { - "id": "deepseek/deepseek-r1-0528-qwen3-8b", - "name": "DeepSeek R1 0528 Qwen3 8B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.6, - "output": 0.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 32000, - "supportsTools": false - }, - "deepseek/deepseek-r1-distill-llama-70b": { - "id": "deepseek/deepseek-r1-distill-llama-70b", - "name": "DeepSeek R1 Distill LLama 70B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 8, - "output": 8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": false, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-r1-turbo": { - "id": "deepseek/deepseek-r1-turbo", - "name": "DeepSeek R1 (Turbo)", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 7, - "output": 25, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 64000, - "maxTokens": 16000, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-r1/community": { - "id": "deepseek/deepseek-r1/community", - "name": "DeepSeek R1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 40, - "output": 40, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 64000, - "maxTokens": 8000, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-v3-0324": { - "id": "deepseek/deepseek-v3-0324", - "name": "DeepSeek V3 0324", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 2.7, - "output": 11.2, - "cacheRead": 1.35, - "cacheWrite": 0 - }, - "contextWindow": 163840, - "maxTokens": 65536, - "supportsTools": true - }, - "deepseek/deepseek-v3-turbo": { - "id": "deepseek/deepseek-v3-turbo", - "name": "DeepSeek V3 (Turbo)", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 4, - "output": 13, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 64000, - "maxTokens": 16000, - "supportsTools": true - }, - "deepseek/deepseek-v3.1": { - "id": "deepseek/deepseek-v3.1", - "name": "DeepSeek V3.1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 2.7, - "output": 10, - "cacheRead": 1.35, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-v3.1-terminus": { - "id": "deepseek/deepseek-v3.1-terminus", - "name": "DeepSeek V3.1 Terminus", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 2.7, - "output": 10, - "cacheRead": 1.35, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-v3.2": { - "id": "deepseek/deepseek-v3.2", - "name": "DeepSeek V3.2", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 2.69, - "output": 4, - "cacheRead": 1.345, - "cacheWrite": 0 - }, - "contextWindow": 163840, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-v3.2-exp": { - "id": "deepseek/deepseek-v3.2-exp", - "name": "DeepSeek V3.2 Exp", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 2.7, - "output": 4.1, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 163840, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-v3/community": { - "id": "deepseek/deepseek-v3/community", - "name": "DeepSeek V3", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 8.9, - "output": 8.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 64000, - "maxTokens": 8000, - "supportsTools": true - }, - "deepseek/deepseek-v4-flash": { - "id": "deepseek/deepseek-v4-flash", - "name": "DeepSeek V4 Flash", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1.4, - "output": 2.8, - "cacheRead": 0.28, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 393216, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "deepseek/deepseek-v4-pro": { - "id": "deepseek/deepseek-v4-pro", - "name": "DeepSeek V4 Pro", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 16, - "output": 32, - "cacheRead": 1.35, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 393216, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "dev/glm46": { - "id": "dev/glm46", - "name": "dev/glm46", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 256000, - "maxTokens": 256000, - "supportsTools": true - }, - "google/gemma-3-12b-it": { - "id": "google/gemma-3-12b-it", - "name": "Gemma3 12B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.5, - "output": 1, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 8192, - "supportsTools": false - }, - "google/gemma-3-27b-it": { - "id": "google/gemma-3-27b-it", - "name": "Gemma 3 27B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.19, - "output": 2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 98304, - "maxTokens": 16384, - "supportsTools": false - }, - "google/gemma-4-26b-a4b-it": { - "id": "google/gemma-4-26b-a4b-it", - "name": "Gemma 4 26B A4B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.3, - "output": 4, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "google/gemma-4-31b-it": { - "id": "google/gemma-4-31b-it", - "name": "Gemma 4 31B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.4, - "output": 4, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "gryphe/mythomax-l2-13b": { - "id": "gryphe/mythomax-l2-13b", - "name": "Mythomax L2 13B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.9, - "output": 0.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 4096, - "maxTokens": 3200, - "supportsTools": false - }, - "gt-4p": { - "id": "gt-4p", - "name": "gt-4p", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": null, - "maxTokens": 131072, - "supportsTools": true - }, - "inclusionai/ling-2.6-1t": { - "id": "inclusionai/ling-2.6-1t", - "name": "Ling-2.6-1T", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 25, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 32768, - "supportsTools": true - }, - "inclusionai/ling-2.6-flash": { - "id": "inclusionai/ling-2.6-flash", - "name": "Ling-2.6 Flash", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.2, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 32768, - "supportsTools": true - }, - "inclusionai/ring-2.6-1t": { - "id": "inclusionai/ring-2.6-1t", - "name": "Ring-2.6-1T", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 25, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "kwaipilot/kat-coder-pro": { - "id": "kwaipilot/kat-coder-pro", - "name": "Kat Coder Pro", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 12, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 256000, - "maxTokens": 128000, - "supportsTools": true - }, - "meta-llama/llama-3.1-8b-instruct": { - "id": "meta-llama/llama-3.1-8b-instruct", - "name": "Llama 3.1 8B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.2, - "output": 0.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 16384, - "maxTokens": 16384, - "supportsTools": false - }, - "meta-llama/llama-3.2-1b-instruct": { - "id": "meta-llama/llama-3.2-1b-instruct", - "name": "Llama 3.2 1B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.2, - "output": 0.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131000, - "maxTokens": 32000, - "supportsTools": false - }, - "meta-llama/llama-3.2-3b-instruct": { - "id": "meta-llama/llama-3.2-3b-instruct", - "name": "Llama 3.2 3B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.3, - "output": 0.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 32768, - "maxTokens": 32000, - "supportsTools": false - }, - "meta-llama/llama-3.3-70b-instruct": { - "id": "meta-llama/llama-3.3-70b-instruct", - "name": "Llama 3.3 70B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 1.35, - "output": 4, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 6000, - "maxTokens": 120000, - "supportsTools": true - }, - "meta-llama/llama-4-maverick-17b-128e-instruct-fp8": { - "id": "meta-llama/llama-4-maverick-17b-128e-instruct-fp8", - "name": "Llama 4 Maverick Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.7, - "output": 8.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 8192, - "supportsTools": false - }, - "meta-llama/llama-4-scout-17b-16e-instruct": { - "id": "meta-llama/llama-4-scout-17b-16e-instruct", - "name": "Llama 4 Scout 17B 16E", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.8, - "output": 5.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 131072, - "supportsTools": false - }, - "microsoft/wizardlm-2-8x22b": { - "id": "microsoft/wizardlm-2-8x22b", - "name": "Wizardlm 2 8x22B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 6.2, - "output": 6.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 65535, - "maxTokens": 8000, - "supportsTools": false - }, - "minimax/m2-her": { - "id": "minimax/m2-her", - "name": "M2-her", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 32000, - "maxTokens": null, - "supportsTools": false - }, - "minimax/minimax-m2": { - "id": "minimax/minimax-m2", - "name": "MiniMax M2", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 12, - "cacheRead": 0.3, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "minimax/minimax-m2.1": { - "id": "minimax/minimax-m2.1", - "name": "MiniMax M2.1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 12, - "cacheRead": 0.3, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "minimax/minimax-m2.5": { - "id": "minimax/minimax-m2.5", - "name": "MiniMax M2.5", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 12, - "cacheRead": 0.3, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131100, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "minimax/minimax-m2.5-highspeed": { - "id": "minimax/minimax-m2.5-highspeed", - "name": "MiniMax M2.5-highspeed", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 6, - "output": 24, - "cacheRead": 0.3, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131100, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "minimax/minimax-m2.7": { - "id": "minimax/minimax-m2.7", - "name": "MiniMax M2.7", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 12, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "minimax/minimax-m2.7-highspeed": { - "id": "minimax/minimax-m2.7-highspeed", - "name": "MiniMax M2.7 highspeed", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 6, - "output": 24, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "minimax/minimax-m3": { - "id": "minimax/minimax-m3", - "name": "MiniMax M3", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 12, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "minimaxai/minimax-m1-80k": { - "id": "minimaxai/minimax-m1-80k", - "name": "MiniMax M1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 5.5, - "output": 22, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 40000, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "mistralai/mistral-nemo": { - "id": "mistralai/mistral-nemo", - "name": "Mistral Nemo", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.4, - "output": 1.7, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 60288, - "maxTokens": 16000, - "supportsTools": false - }, - "moonshotai/kimi-k2-0905": { - "id": "moonshotai/kimi-k2-0905", - "name": "Kimi K2 0905", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 6, - "output": 25, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 100352, - "supportsTools": true - }, - "moonshotai/kimi-k2-instruct": { - "id": "moonshotai/kimi-k2-instruct", - "name": "Kimi K2 Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 5.7, - "output": 23, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 100352, - "supportsTools": true - }, - "moonshotai/kimi-k2-thinking": { - "id": "moonshotai/kimi-k2-thinking", - "name": "Kimi K2 Thinking", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 6, - "output": 25, - "cacheRead": 1.5, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 100352, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "requiresEffort": true - } - }, - "moonshotai/kimi-k2.5": { - "id": "moonshotai/kimi-k2.5", - "name": "Kimi K2.5", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 6, - "output": 30, - "cacheRead": 1, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 262144, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "moonshotai/kimi-k2.6": { - "id": "moonshotai/kimi-k2.6", - "name": "Kimi K2.6", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 8, - "output": 34, - "cacheRead": 1.6, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 262144, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "moonshotai/kimi-k2.7-code": { - "id": "moonshotai/kimi-k2.7-code", - "name": "Kimi K2.7 Code", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 9.5, - "output": 40, - "cacheRead": 1.9, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 262144, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "nousresearch/hermes-2-pro-llama-3-8b": { - "id": "nousresearch/hermes-2-pro-llama-3-8b", - "name": "Hermes 2 Pro Llama 3 8B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 1.4, - "output": 1.4, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": false - }, - "nvidia/nemotron-3-nano-30b-a3b": { - "id": "nvidia/nemotron-3-nano-30b-a3b", - "name": "Nemotron 3 Nano 30B A3B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.5, - "output": 2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "openai/gpt-oss-120b": { - "id": "openai/gpt-oss-120b", - "name": "GPT OSS 120B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.5, - "output": 2.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ] - } - }, - "openai/gpt-oss-20b": { - "id": "openai/gpt-oss-20b", - "name": "GPT OSS 20B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.4, - "output": 1.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": false, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ] - } - }, - "paddlepaddle/paddleocr-vl": { - "id": "paddlepaddle/paddleocr-vl", - "name": "PaddleOCR-VL", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.2, - "output": 0.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 16384, - "maxTokens": 16384, - "supportsTools": false - }, - "qwen/qwen-2.5-72b-instruct": { - "id": "qwen/qwen-2.5-72b-instruct", - "name": "Qwen2.5 72B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 3.8, - "output": 4, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 32000, - "maxTokens": 8192, - "supportsTools": true - }, - "qwen/qwen-mt-plus": { - "id": "qwen/qwen-mt-plus", - "name": "Qwen MT Plus", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 2.5, - "output": 7.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 16384, - "maxTokens": 8192, - "supportsTools": false - }, - "qwen/qwen3-235b-a22b-fp8": { - "id": "qwen/qwen3-235b-a22b-fp8", - "name": "Qwen3 235B A22B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 2, - "output": 8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 40960, - "maxTokens": 20000, - "supportsTools": false, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3-235b-a22b-instruct-2507": { - "id": "qwen/qwen3-235b-a22b-instruct-2507", - "name": "Qwen3 235B A22B Instruct 2507", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.9, - "output": 5.8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 16384, - "supportsTools": true - }, - "qwen/qwen3-235b-a22b-thinking-2507": { - "id": "qwen/qwen3-235b-a22b-thinking-2507", - "name": "Qwen3 235B A22B Thinking 2507", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 30, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "qwen/qwen3-coder-30b-a3b-instruct": { - "id": "qwen/qwen3-coder-30b-a3b-instruct", - "name": "Qwen3 Coder 30B A3B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.7, - "output": 2.7, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 160000, - "maxTokens": 32768, - "supportsTools": true - }, - "qwen/qwen3-coder-480b-a35b-instruct": { - "id": "qwen/qwen3-coder-480b-a35b-instruct", - "name": "Qwen3 Coder 480B A35B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 3.8, - "output": 15.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true - }, - "qwen/qwen3-coder-next": { - "id": "qwen/qwen3-coder-next", - "name": "Qwen3 Coder Next", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 2, - "output": 15, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true - }, - "qwen/qwen3-max": { - "id": "qwen/qwen3-max", - "name": "Qwen3 Max", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 21.1, - "output": 84.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true - }, - "qwen/qwen3-next-80b-a3b-instruct": { - "id": "qwen/qwen3-next-80b-a3b-instruct", - "name": "Qwen3-Next-80B-A3B-Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 1.5, - "output": 15, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true - }, - "qwen/qwen3-omni-30b-a3b-instruct": { - "id": "qwen/qwen3-omni-30b-a3b-instruct", - "name": "Qwen3 Omni 30B A3B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.5, - "output": 9.7, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 65536, - "maxTokens": 16384, - "supportsTools": true - }, - "qwen/qwen3-omni-30b-a3b-thinking": { - "id": "qwen/qwen3-omni-30b-a3b-thinking", - "name": "Qwen3 Omni 30B A3B Thinking", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.5, - "output": 9.7, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 65536, - "maxTokens": 16384, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "qwen/qwen3-vl-235b-a22b-instruct": { - "id": "qwen/qwen3-vl-235b-a22b-instruct", - "name": "Qwen3 VL 235B A22B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 15, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true - }, - "qwen/qwen3-vl-235b-a22b-thinking": { - "id": "qwen/qwen3-vl-235b-a22b-thinking", - "name": "Qwen3 VL 235B A22B Thinking", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 9.8, - "output": 39.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "qwen/qwen3-vl-30b-a3b-instruct": { - "id": "qwen/qwen3-vl-30b-a3b-instruct", - "name": "Qwen3 VL 30B A3B Instruct", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2, - "output": 7, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true - }, - "qwen/qwen3.5-122b-a10b": { - "id": "qwen/qwen3.5-122b-a10b", - "name": "Qwen3.5 122B-A10B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 4, - "output": 32, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.5-27b": { - "id": "qwen/qwen3.5-27b", - "name": "Qwen3.5-27B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 24, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.5-35b-a3b": { - "id": "qwen/qwen3.5-35b-a3b", - "name": "Qwen3.5-35B-A3B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.5, - "output": 20, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.5-397b-a17b": { - "id": "qwen/qwen3.5-397b-a17b", - "name": "Qwen3.5 397B A17B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 6, - "output": 36, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.5-plus": { - "id": "qwen/qwen3.5-plus", - "name": "Qwen3.5 Plus", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.6-27b": { - "id": "qwen/qwen3.6-27b", - "name": "Qwen3.6 27B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 6, - "output": 36, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.6-35b-a3b": { - "id": "qwen/qwen3.6-35b-a3b", - "name": "Qwen3.6-35B-A3B", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.48, - "output": 14.85, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.6-plus": { - "id": "qwen/qwen3.6-plus", - "name": "Qwen3.6 Plus", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "qwen/qwen3.7-max": { - "id": "qwen/qwen3.7-max", - "name": "Qwen3.7 Max", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 12.5, - "output": 37.5, - "cacheRead": 2.5, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 65536, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "sao10k/l3-70b-euryale-v2.1": { - "id": "sao10k/l3-70b-euryale-v2.1", - "name": "L3 70B Euryale V2.1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 14.8, - "output": 14.8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": true - }, - "sao10k/l3-8b-lunaris": { - "id": "sao10k/l3-8b-lunaris", - "name": "Sao10k L3 8B Lunaris", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.5, - "output": 0.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": false - }, - "Sao10K/L3-8B-Stheno-v3.2": { - "id": "Sao10K/L3-8B-Stheno-v3.2", - "name": "L3 8B Stheno V3.2", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.5, - "output": 0.5, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 32000, - "supportsTools": true - }, - "sao10k/l31-70b-euryale-v2.2": { - "id": "sao10k/l31-70b-euryale-v2.2", - "name": "L31 70B Euryale V2.2", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 14.8, - "output": 14.8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 8192, - "maxTokens": 8192, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "stepfun/step-3.7-flash": { - "id": "stepfun/step-3.7-flash", - "name": "Step 3.7 Flash", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2, - "output": 11.5, - "cacheRead": 0.4, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 256000, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "tencent/hy3": { - "id": "tencent/hy3", - "name": "Hy3", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 262144, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "thudm/glm-4-32b-0414": { - "id": "thudm/glm-4-32b-0414", - "name": "GLM-4-32B-0414", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 5.5, - "output": 16.6, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 32000, - "maxTokens": 32000, - "supportsTools": true - }, - "xiaomimimo/mimo-v2.5": { - "id": "xiaomimimo/mimo-v2.5", - "name": "XiaomiMiMo/MiMo-V2.5", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.68, - "output": 3.36, - "cacheRead": 0.034, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ] - } - }, - "xiaomimimo/mimo-v2.5-pro": { - "id": "xiaomimimo/mimo-v2.5-pro", - "name": "XiaomiMiMo/MiMo-V2.5-Pro", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 5.22, - "output": 10.44, - "cacheRead": 0.043, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ] - } - }, - "zai-org/autoglm-phone-9b-multilingual": { - "id": "zai-org/autoglm-phone-9b-multilingual", - "name": "AutoGLM-Phone-9B-Multilingual", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.35, - "output": 1.38, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 65536, - "maxTokens": 65536, - "supportsTools": false - }, - "zai-org/glm-4.5-air": { - "id": "zai-org/glm-4.5-air", - "name": "zai-org/glm-4.5-air", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1.3, - "output": 8.5, - "cacheRead": 0.25, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 98304, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-4.5v": { - "id": "zai-org/glm-4.5v", - "name": "GLM 4.5V", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 6, - "output": 18, - "cacheRead": 1.1, - "cacheWrite": 0 - }, - "contextWindow": 65536, - "maxTokens": 16384, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-4.6": { - "id": "zai-org/glm-4.6", - "name": "GLM 4.6", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 5.5, - "output": 22, - "cacheRead": 1.1, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-4.6v": { - "id": "zai-org/glm-4.6v", - "name": "GLM 4.6V", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 9, - "cacheRead": 0.55, - "cacheWrite": 0 - }, - "contextWindow": 131072, - "maxTokens": 32768, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-4.7": { - "id": "zai-org/glm-4.7", - "name": "GLM 4.7", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 6, - "output": 22, - "cacheRead": 1.1, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-4.7-flash": { - "id": "zai-org/glm-4.7-flash", - "name": "GLM 4.7 Flash", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.7, - "output": 4, - "cacheRead": 0.1, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 128000, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-4.7-h": { - "id": "zai-org/glm-4.7-h", - "name": "GLM-4.7", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 6, - "output": 22, - "cacheRead": 1.1, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-5": { - "id": "zai-org/glm-5", - "name": "GLM 5", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 10, - "output": 32, - "cacheRead": 2, - "cacheWrite": 0 - }, - "contextWindow": 202800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-5-turbo": { - "id": "zai-org/glm-5-turbo", - "name": "GLM-5-Turbo", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 12, - "output": 40, - "cacheRead": 2.4, - "cacheWrite": 0 - }, - "contextWindow": 202800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-5.1": { - "id": "zai-org/glm-5.1", - "name": "GLM 5.1", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 13.8, - "output": 44, - "cacheRead": 2.6, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "zai-org/glm-5.2": { - "id": "zai-org/glm-5.2", - "name": "GLM 5.2", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 14, - "output": 44, - "cacheRead": 2.6, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } - } - }, - "zai-org/glm-5v-turbo": { - "id": "zai-org/glm-5v-turbo", - "name": "GLM-5V-Turbo", - "api": "openai-completions", - "provider": "novita", - "baseUrl": "https://api.novita.ai/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 12, - "output": 40, - "cacheRead": 2.4, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "supportsTools": true, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - } - }, "ollama-cloud": { "cogito-2.1:671b": { "id": "cogito-2.1:671b", diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index f993a6538..57c169220 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -277,6 +277,7 @@ export const CATALOG_PROVIDERS = [ defaultModel: "moonshotai/kimi-k2.7-code", envVars: ["NOVITA_API_KEY"], createModelManagerOptions: (config: ModelManagerConfig) => novitaModelManagerOptions(config), + dynamicModelsAuthoritative: true, catalogDiscovery: { label: "Novita", allowUnauthenticated: true }, }, { diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 2e8cd4438..0b785b6ca 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -975,6 +975,7 @@ export function nvidiaModelManagerOptions( // 5.5 Novita // --------------------------------------------------------------------------- +/** Novita OpenAI-compatible discovery configuration. */ export interface NovitaModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -989,8 +990,9 @@ function isPublicNovitaModelId(id: string): boolean { return !id.toLowerCase().startsWith("ai_infer_test"); } +// Novita reports token prices in 1/10,000 USD per million tokens. function toNovitaCostPerMillion(value: unknown): number { - return toPositiveNumber(value, 0) / 1000; + return toPositiveNumber(value, 0) / 10_000; } function getNovitaCacheReadPricePerMillion(entry: OpenAICompatibleModelRecord): number { @@ -1034,16 +1036,16 @@ function mapNovitaModel( }; } +/** Builds Novita's public model-discovery manager. */ export function novitaModelManagerOptions( config?: NovitaModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { const apiKey = config?.apiKey; const baseUrl = config?.baseUrl ?? "https://api.novita.ai/openai/v1"; - const references = createBundledReferenceMap<"openai-completions">( - "novita" as Parameters[0], - ); + const references = createBundledReferenceMap<"openai-completions">("novita"); return { providerId: "novita", + dynamicModelsAuthoritative: true, fetchDynamicModels: async () => fetchOpenAICompatibleModels({ api: "openai-completions", @@ -1057,7 +1059,7 @@ export function novitaModelManagerOptions( active && isPublicNovitaModelId(model.id) && novitaArrayIncludes(entry.endpoints, "chat/completions") && - model.maxTokens !== 0 + toPositiveNumber(entry.max_output_tokens, 0) > 0 ); }, fetch: config?.fetch, diff --git a/packages/catalog/test/novita-provider.test.ts b/packages/catalog/test/novita-provider.test.ts index f737ec9dc..3bb351d68 100644 --- a/packages/catalog/test/novita-provider.test.ts +++ b/packages/catalog/test/novita-provider.test.ts @@ -1,6 +1,4 @@ import { describe, expect, test } from "bun:test"; -import { getOAuthProviders } from "@oh-my-pi/pi-ai/registry/oauth"; -import { getEnvApiKey } from "@oh-my-pi/pi-ai/stream"; import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "@oh-my-pi/pi-catalog/provider-models/descriptors"; import { novitaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; @@ -11,29 +9,10 @@ describe("Novita built-in provider", () => { expect(descriptor?.defaultModel).toBe("moonshotai/kimi-k2.7-code"); expect(descriptor?.catalogDiscovery?.envVars).toContain("NOVITA_API_KEY"); expect(descriptor?.catalogDiscovery?.allowUnauthenticated).toBe(true); + expect(descriptor?.dynamicModelsAuthoritative).toBe(true); expect(DEFAULT_MODEL_PER_PROVIDER.novita).toBe("moonshotai/kimi-k2.7-code"); }); - test("registers Novita as an API-key login provider", () => { - const provider = getOAuthProviders().find(item => item.id === "novita"); - expect(provider?.name).toBe("Novita"); - expect(provider?.available).toBe(true); - }); - - test("resolves NOVITA_API_KEY via env", () => { - const previous = Bun.env.NOVITA_API_KEY; - Bun.env.NOVITA_API_KEY = "novita-test-key"; - try { - expect(getEnvApiKey("novita")).toBe("novita-test-key"); - } finally { - if (previous === undefined) { - delete Bun.env.NOVITA_API_KEY; - } else { - Bun.env.NOVITA_API_KEY = previous; - } - } - }); - test("maps Novita model catalog metadata from the public OpenAI-compatible endpoint", async () => { const requests: string[] = []; const fetchMock = async (input: string | URL | Request): Promise => { @@ -74,6 +53,23 @@ describe("Novita built-in provider", () => { endpoints: ["chat/completions"], input_modalities: ["text"], }, + { + id: "minimax/m2-her", + status: 1, + context_size: 32000, + features: ["serverless"], + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + { + id: "test/zero-output", + status: 1, + context_size: 32000, + max_output_tokens: 0, + features: ["serverless"], + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, ], }); }; @@ -83,6 +79,7 @@ describe("Novita built-in provider", () => { const model = models?.find(item => item.id === "moonshotai/kimi-k2.7-code"); expect(requests).toEqual(["https://api.novita.ai/openai/v1/models"]); + expect(options.dynamicModelsAuthoritative).toBe(true); expect(models?.map(item => item.id)).toEqual(["moonshotai/kimi-k2.7-code"]); expect(model?.provider).toBe("novita"); expect(model?.baseUrl).toBe("https://api.novita.ai/openai/v1"); @@ -90,7 +87,7 @@ describe("Novita built-in provider", () => { expect(model?.reasoning).toBe(true); expect(model?.supportsTools).toBe(true); expect(model?.input).toEqual(["text", "image"]); - expect(model?.cost).toEqual({ input: 9.5, output: 40, cacheRead: 1.9, cacheWrite: 0 }); + expect(model?.cost).toEqual({ input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 }); expect(model?.contextWindow).toBe(262144); expect(model?.maxTokens).toBe(131072); }); From 68c3c7ea9d92e229e41d278d8b3988495303e961 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:07:07 +0200 Subject: [PATCH 053/205] docs(providers): documented novita support --- .omp/commands/triage.md | 2 +- README.md | 2 +- docs/environment-variables.md | 1 + docs/providers.md | 1 + packages/ai/CHANGELOG.md | 1 + packages/ai/README.md | 7 +++++-- packages/catalog/CHANGELOG.md | 1 + 7 files changed, 11 insertions(+), 4 deletions(-) diff --git a/.omp/commands/triage.md b/.omp/commands/triage.md index 8fa481984..83ac48821 100644 --- a/.omp/commands/triage.md +++ b/.omp/commands/triage.md @@ -72,7 +72,7 @@ For each candidate issue, read the title, body, and **all comments** (comments o | `providers` | Provider-related behavior (generic provider scope) | **Provider labels** (apply only when a specific provider is explicitly involved): -`provider:anthropic`, `provider:bedrock`, `provider:brave`, `provider:cerebras`, `provider:cloudflare`, `provider:codex`, `provider:copilot`, `provider:cursor`, `provider:exa`, `provider:gemini`, `provider:gitlab`, `provider:groq`, `provider:huggingface`, `provider:jina`, `provider:kimi`, `provider:litellm`, `provider:minimax`, `provider:mistral`, `provider:moonshot`, `provider:nanogpt`, `provider:nvidia`, `provider:openai`, `provider:opencode`, `provider:openrouter`, `provider:perplexity`, `provider:qianfan`, `provider:qwen`, `provider:synthetic`, `provider:together`, `provider:venice`, `provider:vercel`, `provider:xai`, `provider:xiaomi`, `provider:zai` +`provider:anthropic`, `provider:bedrock`, `provider:brave`, `provider:cerebras`, `provider:cloudflare`, `provider:codex`, `provider:copilot`, `provider:cursor`, `provider:exa`, `provider:gemini`, `provider:gitlab`, `provider:groq`, `provider:huggingface`, `provider:jina`, `provider:kimi`, `provider:litellm`, `provider:minimax`, `provider:mistral`, `provider:moonshot`, `provider:nanogpt`, `provider:novita`, `provider:nvidia`, `provider:openai`, `provider:opencode`, `provider:openrouter`, `provider:perplexity`, `provider:qianfan`, `provider:qwen`, `provider:synthetic`, `provider:together`, `provider:venice`, `provider:vercel`, `provider:xai`, `provider:xiaomi`, `provider:zai` **Platform labels** (apply only when platform materially affects reproduction/root cause): | Label | Signals | diff --git a/README.md b/README.md index 15e7a878e..ec9f6074f 100644 --- a/README.md +++ b/README.md @@ -292,7 +292,7 @@ Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google A Subscription-routed. `/login` attaches the session. -Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen +Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Novita · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen ### Run it yourself diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 9d323aad7..adf3d310c 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -49,6 +49,7 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not | `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | | | `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | | | `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | | +| `NOVITA_API_KEY` | Novita auth | Using `novita` provider | | | `VENICE_API_KEY` | Venice auth | Using `venice` provider | | | `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key | | `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required | diff --git a/docs/providers.md b/docs/providers.md index c5f65e009..cb2b4ec53 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -105,6 +105,7 @@ Each provider has one or more environment variables that supply a key when no st | `huggingface` | `HUGGINGFACE_HUB_TOKEN`, then `HF_TOKEN` | | `moonshot` | `MOONSHOT_API_KEY` | | `nanogpt` | `NANO_GPT_API_KEY` | +| `novita` | `NOVITA_API_KEY` | | `venice` | `VENICE_API_KEY` | | `vercel-ai-gateway` | `AI_GATEWAY_API_KEY` (also `VERCEL_AI_GATEWAY_API_KEY` for catalog discovery) | | `cloudflare-ai-gateway` | `CLOUDFLARE_AI_GATEWAY_API_KEY` | diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 05e484348..6c7954447 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -7,6 +7,7 @@ - Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. - Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`. - Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress. +- Added Novita API-key login with authenticated key validation and `NOVITA_API_KEY` discovery ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). ### Changed diff --git a/packages/ai/README.md b/packages/ai/README.md index 156baf4a4..bb47da22e 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -59,6 +59,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an - **Qianfan** (requires `QIANFAN_API_KEY`) - **NVIDIA** (requires `NVIDIA_API_KEY`) - **NanoGPT** (requires `NANO_GPT_API_KEY`) +- **Novita** (requires `NOVITA_API_KEY`) - **Hugging Face Inference** - **xAI** - **Venice** (requires `VENICE_API_KEY`) @@ -943,6 +944,7 @@ In Node.js environments, you can set environment variables to avoid passing API | Synthetic | `SYNTHETIC_API_KEY` | | NVIDIA | `NVIDIA_API_KEY` | | NanoGPT | `NANO_GPT_API_KEY` | +| Novita | `NOVITA_API_KEY` | | Venice | `VENICE_API_KEY` | | Moonshot | `MOONSHOT_API_KEY` | | xAI | `XAI_API_KEY` | @@ -981,6 +983,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations: - Qianfan: `https://qianfan.baidubce.com/v2` - NVIDIA: `https://integrate.api.nvidia.com/v1` - NanoGPT: `https://nano-gpt.com/api/v1` +- Novita: `https://api.novita.ai/openai/v1` - Hugging Face Inference: `https://router.huggingface.co/v1` - Venice: `https://api.venice.ai/api/v1` - Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic` @@ -1082,7 +1085,7 @@ Credentials are saved to `agent.db` in the agent directory. `/login qianfan` ope `login` supports OAuth providers (Anthropic, OpenAI Codex, GitHub Copilot, Gemini CLI, Antigravity) and API-key onboarding flows. -For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth. +For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth. ### Programmatic OAuth @@ -1114,7 +1117,7 @@ import { getOAuthApiKey, // (provider, credentialsMap) => { newCredentials, apiKey } | null // Types - type OAuthProvider, // includes 'anthropic', 'openai-codex', 'github-copilot', 'google-gemini-cli', 'google-antigravity', 'together', 'moonshot', 'qianfan', 'nvidia', 'nanogpt', 'huggingface', 'venice', 'xiaomi', 'vllm', 'litellm', 'cloudflare-ai-gateway', 'qwen-portal', ... + type OAuthProvider, // includes 'anthropic', 'openai-codex', 'github-copilot', 'google-gemini-cli', 'google-antigravity', 'together', 'moonshot', 'qianfan', 'nvidia', 'nanogpt', 'novita', 'huggingface', 'venice', 'xiaomi', 'vllm', 'litellm', 'cloudflare-ai-gateway', 'qwen-portal', ... type OAuthCredentials, } from "@oh-my-pi/pi-ai"; ``` diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 8dfc0bb26..2704465ef 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -8,6 +8,7 @@ - Added support for Dolphin Mistral 24b Venice Edition - Added GLM5.2-Fast model - Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) +- Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). - Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. From 3c325cdb6b50d1b962075207736dc56de0d4a3c4 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:15:24 +0200 Subject: [PATCH 054/205] docs(ai): added unreleased changelog entry for bedrock error classification --- packages/ai/CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 6c7954447..237599771 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -22,6 +22,7 @@ - Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract - Fixed sequential-cutoff Codex reasoning summaries repeating earlier content when atomic summary snapshots are replayed or extended. +- Fixed error classification for typed AWS credential-resolution failures (`AwsCredentialsError`) to map them to authentication failures. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) ## [16.3.15] - 2026-07-09 From 70754dfa015e1e3be537ba52caaa6bfb11a73337 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:33:34 +0200 Subject: [PATCH 055/205] fix(coding-agent): hardened same-realm runtime guards from PR review - setCwd now updates the saved __omp_session__ stack entry so a deferred cross-runtime setCwd is visible to the runtime's next run (review should-fix) - JsRuntime installation asserts realm ownership before mutating globals; a first init during another runtime's live run fails via init-failed instead of clobbering the active run's globals - cmux runCmuxCode marks the armed cancel rejection as handled so a sync setup throw under an already-aborted signal cannot become an unhandled rejection (review P2) - credited #4907 in the changelog entry --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/eval/js/shared/runtime.ts | 32 ++++----- .../src/tools/browser/cmux/cmux-tab.ts | 5 ++ .../test/eval/runtime-global-dispose.test.ts | 8 +-- .../test/eval/worker-core.test.ts | 67 +++++++++++++++++++ 5 files changed, 94 insertions(+), 20 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 33d84caac..e8c47ac3d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed interactive TUI sessions dying with `Unhandled rejection: Cannot set cwd while another same-realm JS runtime is running` after the JS eval worker fell back to the in-process inline path (commonly when the worker could not load `pi_natives`). Concurrent inline eval/browser runtimes now stamp cwd without stealing the exclusive realm; exclusive activation remains on `run`/`setRunScope`, and WorkerCore `init` reports failures via `init-failed` instead of throwing out of the microtask path. +- Fixed interactive TUI sessions dying with `Unhandled rejection: Cannot set cwd while another same-realm JS runtime is running` after the JS eval worker fell back to the in-process inline path (commonly when the worker could not load `pi_natives`). Concurrent inline eval/browser runtimes now stamp cwd (including the saved `__omp_session__` state) without stealing the exclusive realm, WorkerCore `init` reports failures via `init-failed` instead of throwing out of the microtask path, and constructing a runtime while another same-realm run is live fails explicitly instead of clobbering its globals. ([#4907](https://github.com/can1357/oh-my-pi/pull/4907) by [@cexll](https://github.com/cexll)) - Fixed compaction aborting instead of trying an authenticated fallback model when Amazon Bedrock credential resolution fails before a request is sent. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) - Fixed full-context forks cold-missing OpenAI prompt caches by persisting an inherited provider prompt-cache key separately from the new OMP session id, adding `--prompt-cache-key` for explicit cache affinity, and dropping automatic inheritance when startup changes the model, thinking level, system prompt, or tool schema. ([#5035](https://github.com/can1357/oh-my-pi/issues/5035)) - Fixed Codex advisor requests using local `-advisor` session labels as provider session IDs; advisors now use stable UUIDv7 provider identities while keeping labeled transcript names. ([#5040](https://github.com/can1357/oh-my-pi/issues/5040)) diff --git a/packages/coding-agent/src/eval/js/shared/runtime.ts b/packages/coding-agent/src/eval/js/shared/runtime.ts index 6af513780..1f9771d35 100644 --- a/packages/coding-agent/src/eval/js/shared/runtime.ts +++ b/packages/coding-agent/src/eval/js/shared/runtime.ts @@ -163,6 +163,7 @@ export class JsRuntime { readonly helpers: HelperBundle; #cwd: string; + #session: { cwd: string; sessionId: string }; readonly sessionId: string; #env: Map; #als = new AsyncLocalStorage(); @@ -171,6 +172,7 @@ export class JsRuntime { constructor(opts: RuntimeOptions) { this.#cwd = opts.initialCwd; + this.#session = { cwd: opts.initialCwd, sessionId: opts.sessionId }; this.sessionId = opts.sessionId; this.#env = new Map(); this.#moduleLoader = new LocalModuleLoader(this.sessionId); @@ -189,24 +191,19 @@ export class JsRuntime { } setCwd(cwd: string): void { - // Always stamp the local field: WorkerCore/browser/cmux call setCwd from - // init and pre-run paths that may race another same-realm runtime. The - // exclusive global bag is only needed when this runtime is about to - // execute; run()/setRunScope still assert ownership. A throw here used - // to escape via the inline-worker microtask path as a fatal - // unhandledRejection and kill the whole interactive session. if (this.#disposed) throw new Error("Cannot set cwd on a disposed JS runtime"); + // Always stamp the runtime and session state: WorkerCore/browser/cmux call + // setCwd from init and pre-run paths that may race another same-realm + // runtime, and a throw here used to escape the inline-worker microtask + // path as a fatal unhandledRejection that killed the whole session. + // #session is the same object saved in this owner's global stack entry, + // so the new cwd survives deferred activation and is visible to this + // runtime's next run; run()/setRunScope still assert exclusive ownership. this.#cwd = cwd; - try { + this.#session.cwd = cwd; + if (activeGlobalRunOwner === null || activeGlobalRunOwner === this.#globalOwner) { this.#activateGlobals("set cwd"); - } catch (err) { - if (err instanceof Error && err.message.includes("another same-realm JS runtime is running")) { - return; - } - throw err; } - const session = (globalThis as { __omp_session__?: { cwd?: string } }).__omp_session__; - if (session) session.cwd = cwd; } /** @@ -332,8 +329,13 @@ export class JsRuntime { } #install(extraGlobals: Record | undefined): void { + // Constructing a runtime while another same-realm runtime is mid-run would + // silently replace the live runtime's globals (Object.assign + prelude eval + // below). Fail before any global/stack mutation; WorkerCore reports it as + // init-failed instead of corrupting the active run. + assertCanUseGlobalOwner(this.#globalOwner, "initialize a JS runtime"); const injected: Record = { - __omp_session__: { cwd: this.#cwd, sessionId: this.sessionId }, + __omp_session__: this.#session, __omp_helpers__: this.helpers, __omp_call_tool__: async (name: string, args: unknown) => { const hooks = this.#activeHooks("tool"); diff --git a/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts b/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts index 5a25bac66..a0e311d29 100644 --- a/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts +++ b/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts @@ -1290,6 +1290,11 @@ export async function runCmuxCode(tab: CmuxTab, opts: RunCmuxCodeOptions): Promi tab.setRunContext({ session: opts.snapshot, displays, screenshots, signal, timeoutMs: opts.timeoutMs }); const { promise: cancelRejection, reject } = Promise.withResolvers(); + // If the synchronous setup below throws (same-realm ownership conflict) + // while `signal` is already aborted, `Promise.race` never attaches a + // handler to this promise; keep its armed rejection from surfacing as an + // unhandled rejection — the postmortem-fatal path this run guards against. + cancelRejection.catch(() => {}); const onAbort = (): void => { if (timeoutSignal.aborted) { reject(new ToolError(`Browser code execution timed out after ${opts.timeoutMs}ms`)); diff --git a/packages/coding-agent/test/eval/runtime-global-dispose.test.ts b/packages/coding-agent/test/eval/runtime-global-dispose.test.ts index 5bf906630..f14f11a16 100644 --- a/packages/coding-agent/test/eval/runtime-global-dispose.test.ts +++ b/packages/coding-agent/test/eval/runtime-global-dispose.test.ts @@ -126,12 +126,12 @@ describe("JsRuntime global disposal", () => { ); gate.resolve(); await activeSecond; - // After the exclusive run ends, setCwd can promote globals and keep the stamped cwd. - first.setCwd(pendingCwd); + // The deferred cwd must reach this runtime's next run WITHOUT a second + // setCwd: the saved __omp_session__ stack entry carries the new value. + expect(await first.run("__omp_session__.cwd", undefined, hooks)).toBe(pendingCwd); expect(first.cwd).toBe(pendingCwd); expect(globals.__omp_helpers__).toBe(first.helpers); - const session = globals.__omp_session__ as { cwd?: string } | undefined; - expect(session?.cwd).toBe(pendingCwd); + expect(globals.__omp_session__).toMatchObject({ cwd: pendingCwd }); } finally { gate.resolve(); if (activeSecond) await activeSecond.catch(() => undefined); diff --git a/packages/coding-agent/test/eval/worker-core.test.ts b/packages/coding-agent/test/eval/worker-core.test.ts index 42cac4cb8..f04cb6352 100644 --- a/packages/coding-agent/test/eval/worker-core.test.ts +++ b/packages/coding-agent/test/eval/worker-core.test.ts @@ -291,6 +291,73 @@ describe("WorkerCore", () => { } }); + it("first init while a same-realm run is live fails via init-failed and recovers", async () => { + const first = createWorkerHarness(); + const second = createWorkerHarness(); // never initialized: no runtime exists yet + const cwd = process.cwd(); + await initializeWorker(first, { cwd, sessionId: "first-init-live-first", localRoots: {} }); + + const gate = Promise.withResolvers(); + const entered = Promise.withResolvers(); + (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }).__omp_worker_core_gate = { + entered: () => entered.resolve(), + wait: gate.promise, + }; + + const { fatal, uninstall } = installFatalCapture(); + try { + const firstText = waitForMessage( + first, + message => message.type === "text" && message.runId === "hold-for-first-init", + ); + const firstResult = waitForMessage( + first, + message => message.type === "result" && message.runId === "hold-for-first-init", + ); + first.send({ + type: "run", + runId: "hold-for-first-init", + code: "globalThis.__omp_worker_core_gate.entered(); await globalThis.__omp_worker_core_gate.wait; __omp_session__.sessionId;", + filename: "[first-init-live-first].js", + snapshot: { cwd, sessionId: "first-init-live-first", localRoots: {} }, + }); + await entered.promise; + + // A fresh runtime's install would Object.assign over the live runtime's + // globals mid-run; it must fail via the protocol instead. + const reply = waitForMessage(second, message => message.type === "ready" || message.type === "init-failed"); + second.send({ type: "init", snapshot: { cwd, sessionId: "first-init-live-second", localRoots: {} } }); + expect(await reply).toMatchObject({ + type: "init-failed", + error: { message: "Cannot initialize a JS runtime while another same-realm JS runtime is running" }, + }); + + // The held run's globals were not clobbered: it still resolves its own + // session bag and completes cleanly. + gate.resolve(); + expect(await firstText).toMatchObject({ + type: "text", + runId: "hold-for-first-init", + chunk: "first-init-live-first\n", + }); + expect(await firstResult).toMatchObject({ type: "result", runId: "hold-for-first-init", ok: true }); + + // Once the realm is free, the same core initializes cleanly. + await initializeWorker(second, { cwd, sessionId: "first-init-live-second", localRoots: {} }); + + // Drain the microtask queue so any latent fatal would surface. + for (let i = 0; i < 8; i++) await Promise.resolve(); + expect(fatal).toEqual([]); + } finally { + uninstall(); + gate.resolve(); + delete (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }) + .__omp_worker_core_gate; + first.send({ type: "close" }); + second.send({ type: "close" }); + } + }); + it("survives concurrent same-realm setCwd in a child process with postmortem loaded", async () => { // Process-level oracle: the production crash was postmortem killing the process // after an unhandled rejection from concurrent inline setCwd. This must stay green From 2bf8f43b6712a8b58338a102df955117591b37a1 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:24:47 +0200 Subject: [PATCH 056/205] fix(agent): keep incremental yields budget-bound --- packages/coding-agent/src/task/executor.ts | 2 +- .../test/task/executor-wall-clock.test.ts | 110 ++++++++++++++++++ 2 files changed, 111 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 9f3ad5a3b..76a9ec463 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -1322,7 +1322,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { } } } - if (softRequestBudget > 0 && !abortSent && !yieldCalled && !yieldCallPending) { + if (softRequestBudget > 0 && !abortSent && !yieldCallPending) { if (progress.requests >= softRequestBudget * 1.5) { requestAbort("budget"); } else if (softRequestBudgetNotice && !budgetSteerSent && progress.requests >= softRequestBudget) { diff --git a/packages/coding-agent/test/task/executor-wall-clock.test.ts b/packages/coding-agent/test/task/executor-wall-clock.test.ts index 6e145fade..0157249ba 100644 --- a/packages/coding-agent/test/task/executor-wall-clock.test.ts +++ b/packages/coding-agent/test/task/executor-wall-clock.test.ts @@ -477,6 +477,116 @@ describe("runSubprocess wall clock (task.maxRuntimeMs)", () => { ]); }); + it("resumes the hard budget guard after an incremental yield commits", async () => { + const settings = Settings.isolated({ "task.softRequestBudget": 1 }); + const firstAssistantMessage = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "still working" }], + stopReason: "stop" as const, + }; + const incrementalYieldMessage = { + role: "assistant" as const, + content: [ + { + type: "toolCall" as const, + id: "tool-yield-incremental", + name: "yield", + arguments: { type: ["findings"], result: { data: { id: "saved" } } }, + }, + ], + stopReason: "toolUse" as const, + }; + const followingAssistantMessage = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "continuing after the saved section" }], + stopReason: "stop" as const, + }; + let listenerRef: ((event: AgentSessionEvent) => void) | undefined; + let lastAssistantMessage: + | typeof firstAssistantMessage + | typeof incrementalYieldMessage + | typeof followingAssistantMessage + | undefined; + let waitForIdleCalls = 0; + let abortCount = 0; + let abortCountBeforeYieldExecutionEnd: number | undefined; + let abortCountAfterFollowingTurn: number | undefined; + const session: Partial = { + state: { messages: [] } as never, + agent: { state: { systemPrompt: ["test"] } } as never, + extensionRunner: undefined as never, + sessionManager: { appendSessionInit: () => {} } as never, + getActiveToolNames: () => ["read", "yield"], + setActiveToolsByName: async () => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + listenerRef = listener; + return () => {}; + }, + prompt: async () => true, + waitForIdle: async () => { + waitForIdleCalls += 1; + if (waitForIdleCalls !== 1) return; + lastAssistantMessage = firstAssistantMessage; + listenerRef?.({ + type: "message_end", + message: firstAssistantMessage, + } as unknown as AgentSessionEvent); + lastAssistantMessage = incrementalYieldMessage; + listenerRef?.({ + type: "message_end", + message: incrementalYieldMessage, + } as unknown as AgentSessionEvent); + abortCountBeforeYieldExecutionEnd = abortCount; + listenerRef?.({ + type: "tool_execution_end", + toolCallId: "tool-yield-incremental", + toolName: "yield", + result: { + content: [{ type: "text", text: "Section submitted." }], + details: { + status: "success", + data: { id: "saved" }, + type: ["findings"], + }, + }, + isError: false, + } as AgentSessionEvent); + lastAssistantMessage = followingAssistantMessage; + listenerRef?.({ + type: "message_end", + message: followingAssistantMessage, + } as unknown as AgentSessionEvent); + abortCountAfterFollowingTurn = abortCount; + }, + getLastAssistantMessage: () => lastAssistantMessage as never, + abort: async () => { + abortCount += 1; + }, + dispose: async () => {}, + }; + mockCreateAgentSession(session as AgentSession); + + const result = await runSubprocess({ + ...baseOptions, + id: "subagent-soft-budget-incremental-yield", + settings, + }); + + expect(abortCountBeforeYieldExecutionEnd).toBe(0); + expect(abortCountAfterFollowingTurn).toBe(1); + expect(result.requests).toBe(3); + expect(result.extractedToolData?.yield).toEqual([ + { + data: { id: "saved" }, + status: "success", + error: undefined, + type: ["findings"], + useLastTurn: undefined, + schemaOverridden: undefined, + }, + ]); + }); + it("propagates per-turn context tokens onto the SingleResult", async () => { // Async task consumers (index.ts) copy `singleResult.contextTokens` and // `singleResult.contextWindow` onto AgentProgress. This test pins the From 0dbb59a14bbd58a8471e600c4b4784894c1740df Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:22:15 +0200 Subject: [PATCH 057/205] test(tui): cover stale cells after destructive repaint --- packages/tui/src/tui.ts | 2 +- packages/tui/test/render-regressions.test.ts | 5 ++++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 955b0e39a..6f60f0db2 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -3162,7 +3162,7 @@ export class TUI extends Container { } /** - * Clear the viewport (optionally native scrollback) and replay the frame: + * Replay the frame from home, optionally clearing native scrollback first: * committed prefix `[0, chunkTo)` followed by the visible window. ED3 * (`CSI 3 J`) is emitted here and only here, and only for gesture-driven * paints (session replace, resize, resetDisplay, or an explicit diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 6d39fa08c..526c73d6d 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -1362,13 +1362,15 @@ describe("TUI terminal-state regressions", () => { setTerminalScreenToScrollback(true); const term = new VirtualTerminal(20, 3); const tui = new TUI(term); - tui.addChild(new MutableLinesComponent(rows("line-", 6))); + const component = new MutableLinesComponent(rows("line-", 6)); + tui.addChild(component); const writes = captureWrites(term); try { tui.start(); await settle(term); writes.length = 0; + component.setLines(["new"]); tui.requestRender(true, { clearScrollback: true }); await settle(term); @@ -1376,6 +1378,7 @@ describe("TUI terminal-state regressions", () => { expect(out).toContain("\x1b[H\x1b[3J"); expect(out).not.toContain("\x1b[2J"); expect(out).not.toContain("\x1b[22J"); + expect(visible(term)).toEqual(["new", "", ""]); } finally { tui.stop(); setTerminalScreenToScrollback(saved); From 390a4ae927aebcc6719716573637b1039cac112a Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:26:51 +0200 Subject: [PATCH 058/205] fix(coding-agent): reconcile late MCP discovery --- packages/coding-agent/src/sdk.ts | 75 +++++++++++-------- .../test/sdk-mcp-auto-discovery.test.ts | 26 +++++++ 2 files changed, 68 insertions(+), 33 deletions(-) diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 6c8667958..16a066675 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1746,37 +1746,17 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } applyMCPEnvironment(mcpResult); logMCPLoadErrors(mcpResult.errors); - // `tools.discoveryMode: "auto"` was resolved against a registry that - // held only built-ins plus persisted placeholder names. Recompute with - // the real MCP tool count: a large toolset must flip discovery on - // BEFORE the refresh, or activateAll would dump every MCP tool into - // the active set with no search_tool_bm25 registered. + // `tools.discoveryMode: "auto"` was resolved before deferred MCP + // tools existed. Reconcile again before refresh so a large toolset + // cannot bypass discovery by arriving after first paint. let discoveryEnabled = activation.mcpDiscoveryEnabled; let activateAll = activation.activateAllMCPTools; - if (!discoveryEnabled) { - const nonMCPToolNames = [...toolRegistry.keys()].filter(name => !isMCPToolName(name)); - const projectedMode = resolveEffectiveToolDiscoveryMode( - settings, - countToolsForAutoDiscovery([...nonMCPToolNames, ...mcpResult.tools.map(tool => tool.name)]), - ); - if (projectedMode !== "off") { - effectiveDiscoveryMode = projectedMode; - mcpDiscoveryEnabled = true; - discoveryEnabled = true; - activateAll = false; - liveSession.enableMCPDiscovery(); - if (!toolRegistry.has("search_tool_bm25")) { - const searchTool: Tool = new SearchToolBm25Tool(toolSession); - toolRegistry.set( - searchTool.name, - new ExtensionToolWrapper(wrapToolWithMetaNotice(searchTool), extensionRunner) as Tool, - ); - } - await liveSession.setActiveToolsByName([ - ...liveSession.getActiveToolNames(), - "search_tool_bm25", - ]); - } + if ( + !discoveryEnabled && + (await enableDeferredMCPDiscoveryForTools(liveSession, mcpResult.tools)) + ) { + discoveryEnabled = true; + activateAll = false; } await liveSession.refreshMCPTools(mcpResult.tools, { activateAll }); if (activation.explicitlyRequestedMCPToolNames.length > 0) { @@ -2317,6 +2297,34 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } let mcpDiscoveryEnabled = effectiveDiscoveryMode !== "off"; // back-compat: true when any discovery active + async function enableDeferredMCPDiscoveryForTools( + liveSession: AgentSession, + mcpTools: CustomTool[], + ): Promise { + if (mcpDiscoveryEnabled) return true; + const nonMCPToolNames = [...toolRegistry.keys()].filter(name => !isMCPToolName(name)); + const projectedMode = resolveEffectiveToolDiscoveryMode( + settings, + countToolsForAutoDiscovery([...nonMCPToolNames, ...mcpTools.map(tool => tool.name)]), + ); + if (projectedMode === "off") return false; + + effectiveDiscoveryMode = projectedMode; + mcpDiscoveryEnabled = true; + liveSession.enableMCPDiscovery(); + if (!toolRegistry.has("search_tool_bm25")) { + const searchTool: Tool = new SearchToolBm25Tool(toolSession); + toolRegistry.set( + searchTool.name, + new ExtensionToolWrapper(wrapToolWithMetaNotice(searchTool), extensionRunner) as Tool, + ); + } + if (!liveSession.getActiveToolNames().includes("search_tool_bm25")) { + await liveSession.setActiveToolsByName([...liveSession.getActiveToolNames(), "search_tool_bm25"]); + } + return true; + } + const reloadSshTool = async (): Promise => { if (!requestedToolNameSet.has("ssh")) return null; const sshTool = (await loadSshTool({ @@ -3075,10 +3083,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} mcpManager.setOnToolsChanged(tools => { void (async () => { try { - await session.refreshMCPTools( - tools, - deferMCPDiscoveryForUI && !mcpDiscoveryEnabled ? { activateAll: true } : undefined, - ); + let activateAll = deferMCPDiscoveryForUI && !mcpDiscoveryEnabled; + if (activateAll && (await enableDeferredMCPDiscoveryForTools(session, tools))) { + activateAll = false; + } + await session.refreshMCPTools(tools, activateAll ? { activateAll: true } : undefined); } catch (error) { logger.warn("MCP tool refresh failed", { error: error instanceof Error ? error.message : String(error), diff --git a/packages/coding-agent/test/sdk-mcp-auto-discovery.test.ts b/packages/coding-agent/test/sdk-mcp-auto-discovery.test.ts index 322747cc8..888c85925 100644 --- a/packages/coding-agent/test/sdk-mcp-auto-discovery.test.ts +++ b/packages/coding-agent/test/sdk-mcp-auto-discovery.test.ts @@ -127,6 +127,32 @@ describe("createAgentSession deferred MCP auto discovery", () => { } }, 40_000); + it("flips auto discovery when MCP tools finish after the startup timeout", async () => { + writeMcpConfig(["--delay", "750"]); + const { session } = await createAgentSession({ ...baseOptions(), toolNames: ["read"] }); + try { + // The manager returns from startup after 250 ms while this fixture is + // still connecting. Its eventual tools arrive through onToolsChanged, + // so wait for that observable registry update rather than a fixed delay. + const deadline = Date.now() + 30_000; + while ( + session.getAllToolNames().filter(name => name.startsWith("mcp__")).length < MANY_TOOL_COUNT && + Date.now() < deadline + ) { + await Bun.sleep(50); + } + + expect(session.isMCPDiscoveryEnabled()).toBe(true); + const activeNames = session.getActiveToolNames(); + expect(activeNames).toContain("read"); + expect(activeNames).toContain("search_tool_bm25"); + expect(activeNames.filter(name => name.startsWith("mcp__"))).toEqual([]); + expect(session.getDiscoverableTools({ source: "mcp" })).toHaveLength(MANY_TOOL_COUNT); + } finally { + await session.dispose(); + } + }, 40_000); + it("disposing mid-connect disconnects the manager and never resurrects tools", async () => { // Stall `initialize` in the real fixture subprocess so the connect is // guaranteed to still be in flight when dispose() runs. Deterministic From 1a72f569356fd507e2377854eab3aa5758be61cc Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:25:27 +0200 Subject: [PATCH 059/205] fix(catalog): keep xAI regeneration scoped --- packages/catalog/src/models.json | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 02e20cc47..b19a10fd5 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -11689,10 +11689,10 @@ "image" ], "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, - "cacheWrite": 3.75 + "input": 2, + "output": 10, + "cacheRead": 0.2, + "cacheWrite": 2.5 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -13686,7 +13686,7 @@ "cacheRead": 0.3, "cacheWrite": 3.75 }, - "contextWindow": 200000, + "contextWindow": 1000000, "maxTokens": 64000, "thinking": { "mode": "budget", From 17c7c6d0f8ef3d2189260d39f79399f1015a499e Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:32:02 +0200 Subject: [PATCH 060/205] fix(agent): preserve external abort boundaries --- packages/agent/src/agent-loop.ts | 13 ++- packages/agent/test/agent-loop.test.ts | 108 +++++++++++++++++- .../coding-agent/src/session/agent-session.ts | 5 +- 3 files changed, 120 insertions(+), 6 deletions(-) diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 9d7f33230..7e32f52f5 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -132,6 +132,13 @@ export function createToolScopedAbortReason( return { kind: "tool-scoped-abort", message, toolCallMessages, defaultToolCallMessage }; } +/** + * Marks an abort raised by a completed post-tool hook as terminal for the + * current run. External/user aborts still synthesize an aborted assistant + * boundary; this reason stops after persisting the completed tool batch. + */ +export const TERMINAL_TOOL_RESULT_ABORT_REASON = Symbol.for("pi-agent-core.terminal-tool-result"); + const STEERING_INTERRUPT_POLL_MS = 250; class HarmonyLeakInterruption extends Error { @@ -1084,9 +1091,9 @@ async function runLoopBody( } } - // A tool hook may abort to mark the tool result as terminal (e.g. subagent yield). - // Stop before the next provider call; event listeners observe the result too late. - if (signal?.aborted) { + // A tool hook may mark its completed result as terminal (e.g. subagent yield). + // Stop before the next provider call without changing external/user abort semantics. + if (signal?.reason === TERMINAL_TOOL_RESULT_ABORT_REASON) { hasMoreToolCalls = false; } diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index 62ec438f4..0ee5f2faa 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -1,5 +1,10 @@ import { describe, expect, it } from "bun:test"; -import { agentLoop, agentLoopContinue, agentLoopDetailed } from "@oh-my-pi/pi-agent-core/agent-loop"; +import { + agentLoop, + agentLoopContinue, + agentLoopDetailed, + TERMINAL_TOOL_RESULT_ABORT_REASON, +} from "@oh-my-pi/pi-agent-core/agent-loop"; import type { AgentContext, AgentEvent, @@ -2353,6 +2358,107 @@ describe("agentLoopContinue with AgentMessage", () => { } }); + it("stops after a post-tool hook marks the completed result terminal", async () => { + const toolSchema = type({ value: "string" }); + const controller = new AbortController(); + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + async execute(_toolCallId, params) { + return { + content: [{ type: "text", text: params.value }], + details: { value: params.value }, + }; + }, + }; + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + const mock = createMockModel({ + responses: [ + { content: [{ type: "toolCall", id: "tool-terminal", name: "echo", arguments: { value: "done" } }] }, + { content: ["must not be reached"] }, + ], + }); + const config: AgentLoopConfig = { + model: mock.model, + convertToLlm: identityConverter, + afterToolCall: async () => { + controller.abort(TERMINAL_TOOL_RESULT_ABORT_REASON); + }, + }; + + const events: AgentEvent[] = []; + const stream = agentLoop([createUserMessage("echo")], context, config, controller.signal, mock.stream); + for await (const event of stream) events.push(event); + + expect(mock.calls).toHaveLength(1); + expect(events.some(event => event.type === "tool_execution_end")).toBe(true); + expect( + events.some( + event => + event.type === "message_end" && + event.message.role === "assistant" && + event.message.stopReason === "aborted", + ), + ).toBe(false); + }); + + it("preserves an external abort boundary when a completed tool ignores cancellation", async () => { + const toolSchema = type({ value: "string" }); + const controller = new AbortController(); + const started = Promise.withResolvers(); + const release = Promise.withResolvers(); + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + async execute(_toolCallId, params) { + started.resolve(); + await release.promise; + return { + content: [{ type: "text", text: params.value }], + details: { value: params.value }, + }; + }, + }; + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + const mock = createMockModel({ + responses: [ + { content: [{ type: "toolCall", id: "tool-abort", name: "echo", arguments: { value: "done" } }] }, + { content: ["must not be observed"] }, + ], + }); + const config: AgentLoopConfig = { + model: mock.model, + convertToLlm: identityConverter, + }; + + const events: AgentEvent[] = []; + const stream = agentLoop([createUserMessage("echo")], context, config, controller.signal, mock.stream); + const consuming = (async () => { + for await (const event of stream) events.push(event); + })(); + await started.promise; + controller.abort("Stopped by user"); + release.resolve(); + await consuming; + + expect(mock.calls).toHaveLength(2); + const aborted = events.find( + event => + event.type === "message_end" && + event.message.role === "assistant" && + event.message.stopReason === "aborted", + ); + expect(aborted).toBeDefined(); + if (aborted?.type !== "message_end" || aborted.message.role !== "assistant") { + throw new Error("Expected an aborted assistant message"); + } + expect(aborted.message.errorMessage).toBe("Stopped by user"); + }); + it("surfaces afterToolCall errors as a tool error result", async () => { const toolSchema = type({ value: "string" }); const tool: AgentTool = { diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 62ae8a835..3f095f938 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -39,6 +39,7 @@ import { resolveTelemetry, type StreamFn, ThinkingLevel, + TERMINAL_TOOL_RESULT_ABORT_REASON, type ToolChoiceDirective, } from "@oh-my-pi/pi-agent-core"; import { @@ -3717,7 +3718,7 @@ export class AgentSession { const alreadyTerminated = this.#synchronouslyTerminatedYieldToolCallIds.delete(event.toolCallId); if (!alreadyTerminated) { this.#markTerminalYieldToolCall(event.toolCallId); - this.agent.abort(); + this.agent.abort(TERMINAL_TOOL_RESULT_ABORT_REASON); } } @@ -4473,7 +4474,7 @@ export class AgentSession { ) { this.#markTerminalYieldToolCall(ctx.toolCall.id); this.#synchronouslyTerminatedYieldToolCallIds.add(ctx.toolCall.id); - this.agent.abort(); + this.agent.abort(TERMINAL_TOOL_RESULT_ABORT_REASON); } return this.#ttsrAfterToolCall(ctx); } From 5809c1a63774cef87ed061cd79d9cd9d7adde27b Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:40:49 +0200 Subject: [PATCH 061/205] chore(changelog): normalized merged unreleased entries --- packages/ai/CHANGELOG.md | 7 ++----- packages/catalog/CHANGELOG.md | 1 - packages/coding-agent/CHANGELOG.md | 7 +++++-- 3 files changed, 7 insertions(+), 8 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 964bb4d02..7f0c5dc5f 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,9 +2,6 @@ ## [Unreleased] -### Fixed - -- Fixed xAI SuperGrok multi-account rotation when an account returns HTTP 403 `run out of credits` / `personal-team-blocked:spending-limit`. That account-local cap is now classified as a usage limit so `streamSimple` auth-retry and `rotateSessionCredential` switch to a sibling `xai-oauth` credential instead of sticking to the exhausted account. ### Added - Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. @@ -17,17 +14,18 @@ - Refactored Responses Lite transport to move tools and instructions into input items - Updated Responses Lite to force parallel tool calling off and strip image detail - Standardized Responses Lite activation via model-level catalog flags - - Recognized Pro Lite as a paid plan tier for OpenAI Codex models - Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape. ### Fixed +- Fixed xAI SuperGrok multi-account rotation when an account returns HTTP 403 `run out of credits` / `personal-team-blocked:spending-limit`. That account-local cap is now classified as a usage limit so `streamSimple` auth-retry and `rotateSessionCredential` switch to a sibling `xai-oauth` credential instead of sticking to the exhausted account. - Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract - Fixed sequential-cutoff Codex reasoning summaries repeating earlier content when atomic summary snapshots are replayed or extended. - Fixed error classification for typed AWS credential-resolution failures (`AwsCredentialsError`) to map them to authentication failures. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) - Fixed OpenAI-compatible chat-completions streams preserving vLLM-style trailing cached-token usage chunks so `cacheRead` and billable `input` session stats are accurate ([#5022](https://github.com/can1357/oh-my-pi/issues/5022)). - Fixed `xai-oauth/grok-4.5` Responses requests to omit unsupported `reasoning.summary` while preserving the documented `reasoning.effort` payload ([#4998](https://github.com/can1357/oh-my-pi/issues/4998)). +- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks when live usage shows all reported windows recovered ([#4980](https://github.com/can1357/oh-my-pi/issues/4980)). ## [16.3.15] - 2026-07-09 @@ -51,7 +49,6 @@ - Refined account selection logic to correctly identify plan types from account metadata - Fixed OpenAI Codex multi-account routing for GPT-5.6: Sol and Luna requests now prefer Plus-or-higher accounts while Terra remains available to Free/Go accounts; local pro-mode aliases inherit their base model's Codex plan eligibility. - Fixed xAI Grok OAuth login to use xAI's device authorization flow: `/login` now opens the verification URL, displays the device code, and polls for approval instead of asking for a pasted redirect or linking to Hermes Agent documentation. -- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks when live usage shows all reported windows recovered ([#4980](https://github.com/can1357/oh-my-pi/issues/4980)). ## [16.3.14] - 2026-07-09 diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 2704465ef..1df563b14 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -9,7 +9,6 @@ - Added GLM5.2-Fast model - Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) - Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). - - Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. ### Changed diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 80da2de62..abb0c3d5a 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) + ### Fixed - Fixed interactive TUI sessions dying with `Unhandled rejection: Cannot set cwd while another same-realm JS runtime is running` after the JS eval worker fell back to the in-process inline path (commonly when the worker could not load `pi_natives`). Concurrent inline eval/browser runtimes now stamp cwd (including the saved `__omp_session__` state) without stealing the exclusive realm, WorkerCore `init` reports failures via `init-failed` instead of throwing out of the microtask path, and constructing a runtime while another same-realm run is live fails explicitly instead of clobbering its globals. ([#4907](https://github.com/can1357/oh-my-pi/pull/4907) by [@cexll](https://github.com/cexll)) @@ -15,15 +19,14 @@ - Fixed Windows bash tool crashes when an explicit timeout fires while a piped command is still streaming output; the JavaScript fallback now reports the timeout without also aborting the native timeout signal. ([#5021](https://github.com/can1357/oh-my-pi/issues/5021)) - Fixed subagent `yield` tool calls being discarded when the soft request budget hard-aborted the same assistant turn before the yield result event landed. ([#5006](https://github.com/can1357/oh-my-pi/issues/5006)) - Fixed `--tools` filtering in interactive sessions disabling deferred MCP tools; MCP tools discovered from configured servers now stay active when the flag limits only built-in tools. ([#5013](https://github.com/can1357/oh-my-pi/issues/5013)) - - Fixed kept-alive task subagents entering a repeated provider-call loop after an IRC wake and terminal `yield`. ([#4963](https://github.com/can1357/oh-my-pi/issues/4963)) + ## [16.3.15] - 2026-07-09 ### Changed - Integrated testing guidance directly into the main system prompt for improved workflow cohesion - Moved testing guidance into the main system prompt and removed the bundled Tester subagent. -- Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) ## [16.3.14] - 2026-07-09 From 2d2a10ab9d1db17ce86dcef087e6319b3208c604 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:40:49 +0200 Subject: [PATCH 062/205] style(coding-agent): sorted merged agent imports --- packages/coding-agent/src/session/agent-session.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3f095f938..a98f48521 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -38,8 +38,8 @@ import { createToolScopedAbortReason, resolveTelemetry, type StreamFn, - ThinkingLevel, TERMINAL_TOOL_RESULT_ABORT_REASON, + ThinkingLevel, type ToolChoiceDirective, } from "@oh-my-pi/pi-agent-core"; import { From 6dbbfbe1e02a4b8b468606357f67e1634f96b3b9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:50:13 +0200 Subject: [PATCH 063/205] feat(coding-agent): renamed explore agent to scout - Renamed the `explore` agent to `scout` throughout prompt templates, agent definitions, and configuration schemas. - Updated documentation and internal tool references to reflect the new agent identity. --- packages/coding-agent/CHANGELOG.md | 6 ++++++ packages/coding-agent/src/config/settings-schema.ts | 2 +- packages/coding-agent/src/prompts/agents/init.md | 2 +- packages/coding-agent/src/prompts/agents/plan.md | 4 ++-- packages/coding-agent/src/prompts/agents/reviewer.md | 2 +- .../src/prompts/agents/{explore.md => scout.md} | 5 +++-- .../coding-agent/src/prompts/system/plan-mode-active.md | 4 ++-- packages/coding-agent/src/prompts/tools/ast-grep.md | 2 +- packages/coding-agent/src/prompts/tools/grep.md | 2 +- packages/coding-agent/src/prompts/tools/task.md | 4 ++-- packages/coding-agent/src/task/agents.ts | 5 ++--- packages/coding-agent/src/task/executor.ts | 2 +- packages/coding-agent/src/utils/file-display-mode.ts | 2 +- packages/coding-agent/test/acp-lazy-startup.test.ts | 2 +- .../test/agent-session-tool-rebuild-skip.test.ts | 2 +- .../test/tools/task-agent-capabilities.test.ts | 8 ++++---- 16 files changed, 30 insertions(+), 24 deletions(-) rename packages/coding-agent/src/prompts/agents/{explore.md => scout.md} (90%) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index abb0c3d5a..658e3a904 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,8 +2,14 @@ ## [Unreleased] +### Breaking Changes + +- Renamed the bundled agent `explore` to `scout` (including renaming its configuration keys, prompt files, and task definitions). Any configurations, allowlists, or invocations referencing `explore` must now use `scout`. + ### Changed +- Renamed the bundled agent `explore` to `scout` (including all internal references, prompt definitions, and task tool configurations). Any custom configurations or task invocations referencing `explore` must now use `scout`. + - Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) ### Fixed diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index c632057bf..5ee3269d1 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -4148,7 +4148,7 @@ export const SETTINGS_SCHEMA = { group: "Subagents", label: "Soft Subagent Request Budget", description: - "Soft per-subagent request budget (assistant requests per run). Crossing it can inject a steering notice when task.softRequestBudgetNotice is enabled; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/sonic agents use a lower built-in budget.", + "Soft per-subagent request budget (assistant requests per run). Crossing it can inject a steering notice when task.softRequestBudgetNotice is enabled; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled scout/sonic agents use a lower built-in budget.", options: [ { value: "0", label: "Disabled" }, { value: "40", label: "40 requests" }, diff --git a/packages/coding-agent/src/prompts/agents/init.md b/packages/coding-agent/src/prompts/agents/init.md index 7a0a184af..1e8fd85b7 100644 --- a/packages/coding-agent/src/prompts/agents/init.md +++ b/packages/coding-agent/src/prompts/agents/init.md @@ -4,7 +4,7 @@ description: Generate AGENTS.md for current codebase thinking-level: medium --- -Generate AGENTS.md by launching multiple `explore` agents in parallel (via `task` tool) scanning different areas (core src, tests, configs/build, scripts/docs), then synthesize findings into a single file. +Generate AGENTS.md by launching multiple `scout` agents in parallel (via `task` tool) scanning different areas (core src, tests, configs/build, scripts/docs), then synthesize findings into a single file. - **Project Overview**: Brief description of project purpose diff --git a/packages/coding-agent/src/prompts/agents/plan.md b/packages/coding-agent/src/prompts/agents/plan.md index 528a38d42..66ca395b0 100644 --- a/packages/coding-agent/src/prompts/agents/plan.md +++ b/packages/coding-agent/src/prompts/agents/plan.md @@ -2,7 +2,7 @@ name: plan description: Software architect for complex multi-file architectural decisions. NOT for simple tasks, single-file changes, or tasks completable in <5 tool calls. tools: read, grep, glob, bash, lsp, web_search, ast_grep -spawns: explore +spawns: scout model: pi/plan, pi/slow --- @@ -19,7 +19,7 @@ Analyze the codebase and the user's request. Produce a detailed implementation p 4. Identify types, interfaces, contracts 5. Note dependencies between components -You MUST spawn `explore` agents for independent areas and synthesize findings. +You MUST spawn `scout` agents for independent areas and synthesize findings. ## Phase 3: Design 1. List concrete changes (files, functions, types) diff --git a/packages/coding-agent/src/prompts/agents/reviewer.md b/packages/coding-agent/src/prompts/agents/reviewer.md index 262d72dde..edb4e51a2 100644 --- a/packages/coding-agent/src/prompts/agents/reviewer.md +++ b/packages/coding-agent/src/prompts/agents/reviewer.md @@ -2,7 +2,7 @@ name: reviewer description: "Code review specialist for quality/security analysis" tools: read, grep, glob, bash, lsp, web_search, ast_grep -spawns: explore +spawns: scout model: pi/slow output: properties: diff --git a/packages/coding-agent/src/prompts/agents/explore.md b/packages/coding-agent/src/prompts/agents/scout.md similarity index 90% rename from packages/coding-agent/src/prompts/agents/explore.md rename to packages/coding-agent/src/prompts/agents/scout.md index 5ef518c67..9c769ac16 100644 --- a/packages/coding-agent/src/prompts/agents/explore.md +++ b/packages/coding-agent/src/prompts/agents/scout.md @@ -1,10 +1,11 @@ --- -name: explore -description: Fast read-only codebase scout returning compressed context for handoff +name: scout +description: MUST be used for exploratory codebase research, rapid code analysis, and broad pattern searches. Fast read-only scout returning compressed context for handoff. tools: read, grep, glob, web_search model: pi/smol thinking-level: medium read-summarize: false +blocking: true output: properties: summary: diff --git a/packages/coding-agent/src/prompts/system/plan-mode-active.md b/packages/coding-agent/src/prompts/system/plan-mode-active.md index 4f56ba5a9..5056ff0be 100644 --- a/packages/coding-agent/src/prompts/system/plan-mode-active.md +++ b/packages/coding-agent/src/prompts/system/plan-mode-active.md @@ -40,7 +40,7 @@ Write each section together with its body — block ops need a multi-line sectio You eliminate unknowns by discovering facts, not by asking. -- **Discoverable facts** (file locations, current behavior, signatures, configs): you MUST find them yourself with `glob`, `grep`, `read`, or parallel `explore` subagents. Every path, symbol, signature, and behavior the plan states as fact MUST come from something you actually read this session. Anything you could not confirm you mark inline (`unverified — confirm first`); you NEVER present a guess as settled. Ask only when several real candidates survive exploration — then present them with a recommendation. +- **Discoverable facts** (file locations, current behavior, signatures, configs): you MUST find them yourself with `glob`, `grep`, `read`, or parallel `scout` subagents. Every path, symbol, signature, and behavior the plan states as fact MUST come from something you actually read this session. Anything you could not confirm you mark inline (`unverified — confirm first`); you NEVER present a guess as settled. Ask only when several real candidates survive exploration — then present them with a recommendation. - **Preferences and tradeoffs** (intent, UX, scope edges, performance-vs-simplicity): not derivable from code. Surface these early via `{{askToolName}}` with 2–4 mutually exclusive options and a recommended default. Left unanswered → proceed with the default and record it under Assumptions. Every question MUST change the plan or settle a load-bearing choice. Batch them. You NEVER ask what exploration answers, and you NEVER ask filler. @@ -69,7 +69,7 @@ Every question MUST change the plan or settle a load-bearing choice. Batch them. ## Workflow — parallel -1. **Understand** — focus on the request and the code behind it. Launch parallel `explore` subagents (via `task`) when scope spans areas; give each a distinct focus (existing implementations, related components, test patterns). Hunt for reusable code before proposing new. +1. **Understand** — focus on the request and the code behind it. Launch parallel `scout` subagents (via `task`) when scope spans areas; give each a distinct focus (existing implementations, related components, test patterns). Hunt for reusable code before proposing new. 2. **Design** — draft one approach from what you found, weigh tradeoffs briefly, then commit. For large or cross-cutting work you MAY spawn a critique subagent to pressure-test it before committing. 3. **Review** — read the files you intend to touch and confirm the approach holds against the real code; confirm the plan still answers the literal request; use `{{askToolName}}` to close any remaining preference questions. 4. **Write** — write the plan per **Plan contents** below. diff --git a/packages/coding-agent/src/prompts/tools/ast-grep.md b/packages/coding-agent/src/prompts/tools/ast-grep.md index 17da00837..8948637a5 100644 --- a/packages/coding-agent/src/prompts/tools/ast-grep.md +++ b/packages/coding-agent/src/prompts/tools/ast-grep.md @@ -21,5 +21,5 @@ Structural code search via ast-grep. - AVOID repo-root scans — narrow `path` first - Parse issues = query failure, not absence: fix the pattern or tighten `path` before concluding "no matches" -- Broad cross-subsystem exploration: you SHOULD use the Task tool + explore subagent first +- Broad cross-subsystem exploration: you SHOULD use the Task tool + scout subagent first diff --git a/packages/coding-agent/src/prompts/tools/grep.md b/packages/coding-agent/src/prompts/tools/grep.md index cc8c3f312..8eb4b042f 100644 --- a/packages/coding-agent/src/prompts/tools/grep.md +++ b/packages/coding-agent/src/prompts/tools/grep.md @@ -19,5 +19,5 @@ Greps files using regex. - MUST use built-in `grep` for any content search. NEVER shell out to `grep`, `rg`, `ripgrep`, `ag`, `ack`, `git grep`, `awk`, `sed`-for-search, or any CLI search via Bash — not even for one match or a quick check. -- Open-ended search needing multiple rounds? MUST use the Task tool with the explore subagent, NOT chained `grep` calls. +- Open-ended search needing multiple rounds? MUST use the Task tool with the scout subagent, NOT chained `grep` calls. diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md index 4b3ab8a38..88d7a3e31 100644 --- a/packages/coding-agent/src/prompts/tools/task.md +++ b/packages/coding-agent/src/prompts/tools/task.md @@ -11,10 +11,10 @@ Execution blocks your turn: the call only returns once the work is completely fi {{#if ircEnabled}}- **Steering delivery:** Parent-to-subagent IRC is delivered immediately as steering; subagents blocked in `job poll` / `irc wait` do not need to poll separately for it.{{/if}} - **Role matching:** Assign each subagent a specific `role` (e.g. "Security Reviewer", "DB Migrator"). Do not spawn generic workers. - **No overhead:** Each assignment MUST instruct its agent to skip formatters, linters, and project-wide test suites. You will run those once at the end. -- **One-pass agents:** Prefer agents that investigate **and** edit in a single pass; only spin a read-only discovery step (e.g. `explore`) when the affected files are genuinely unknown. +- **One-pass agents:** Prefer agents that investigate **and** edit in a single pass; only spin a read-only discovery step (e.g. `scout`) when the affected files are genuinely unknown. # Inputs -- `agent` (optional): The base agent type to use (e.g., `explore`, `reviewer`). Defaults to `{{defaultAgent}}`{{#if defaultAgentIsGeneric}} (the general-purpose worker){{/if}} — omit it for the default worker instead of passing `agent: "{{defaultAgent}}"`.{{#if allowedAgentsText}} Current spawn policy allows: {{allowedAgentsText}}.{{/if}} +- `agent` (optional): The base agent type to use (e.g., `scout`, `reviewer`). Defaults to `{{defaultAgent}}`{{#if defaultAgentIsGeneric}} (the general-purpose worker){{/if}} — omit it for the default worker instead of passing `agent: "{{defaultAgent}}"`.{{#if allowedAgentsText}} Current spawn policy allows: {{allowedAgentsText}}.{{/if}} {{#if batchEnabled}} - `context`: Shared project state, constraints, and contracts. Applies to the entire batch; do not duplicate this background into individual tasks. - `tasks[]`: Array of subagents to spawn. diff --git a/packages/coding-agent/src/task/agents.ts b/packages/coding-agent/src/task/agents.ts index d9c72f226..4399f2751 100644 --- a/packages/coding-agent/src/task/agents.ts +++ b/packages/coding-agent/src/task/agents.ts @@ -7,13 +7,12 @@ import { Effort } from "@oh-my-pi/pi-ai"; import { parseFrontmatter, prompt } from "@oh-my-pi/pi-utils"; import { parseAgentFields } from "../discovery/helpers"; import designerMd from "../prompts/agents/designer.md" with { type: "text" }; -import exploreMd from "../prompts/agents/explore.md" with { type: "text" }; // Embed agent markdown files at build time import agentFrontmatterTemplate from "../prompts/agents/frontmatter.md" with { type: "text" }; import librarianMd from "../prompts/agents/librarian.md" with { type: "text" }; - import planMd from "../prompts/agents/plan.md" with { type: "text" }; import reviewerMd from "../prompts/agents/reviewer.md" with { type: "text" }; +import scoutMd from "../prompts/agents/scout.md" with { type: "text" }; import taskMd from "../prompts/agents/task.md" with { type: "text" }; import type { AgentDefinition, AgentSource } from "./types"; @@ -41,7 +40,7 @@ function buildAgentContent(def: EmbeddedAgentDef): string { } const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ - { fileName: "explore.md", template: exploreMd }, + { fileName: "scout.md", template: scoutMd }, { fileName: "plan.md", template: planMd }, { fileName: "designer.md", template: designerMd }, { fileName: "reviewer.md", template: reviewerMd }, diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 76a9ec463..a640b658d 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -88,7 +88,7 @@ const MCP_CALL_TIMEOUT_MS = 60_000; * `task.softRequestBudgetNotice`. */ export const SOFT_REQUEST_BUDGET: Record = { - explore: 40, + scout: 40, sonic: 40, default: 90, }; diff --git a/packages/coding-agent/src/utils/file-display-mode.ts b/packages/coding-agent/src/utils/file-display-mode.ts index 6f895199e..894fb982c 100644 --- a/packages/coding-agent/src/utils/file-display-mode.ts +++ b/packages/coding-agent/src/utils/file-display-mode.ts @@ -21,7 +21,7 @@ export interface FileDisplayModeSession { /** * Computes effective line display mode from session settings/env. * Hashline mode takes precedence and implies line-addressed output everywhere. - * Hashlines are suppressed when the edit tool is not available (e.g. explore agents), + * Hashlines are suppressed when the edit tool is not available (e.g. scout agents), * when the caller signals a `raw` read, and when the source is `immutable` * (e.g. internal URLs like artifact://, agent://, memory:// — there is no edit * path that could consume the anchors). Raw output is returned as-is. diff --git a/packages/coding-agent/test/acp-lazy-startup.test.ts b/packages/coding-agent/test/acp-lazy-startup.test.ts index b0e507afe..3bfea3c77 100644 --- a/packages/coding-agent/test/acp-lazy-startup.test.ts +++ b/packages/coding-agent/test/acp-lazy-startup.test.ts @@ -258,7 +258,7 @@ describe("ACP lazy startup", () => { "task.batch": false, "task.maxConcurrency": 4, "task.maxRecursionDepth": 5, - "task.disabledAgents": ["explore"], + "task.disabledAgents": ["scout"], "task.agentModelOverrides": { task: "claude-sonnet-4-20250514" }, "memory.backend": "local", "memories.enabled": true, diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 00b909753..4f14bea98 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -428,7 +428,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { // Mutate the settings-backed state. The tool object identity does not change, // but its `description` getter now returns a new string. The signature must // pick this up live (no per-tool caching) and force a rebuild. - settingState.disabled = "plan,explore"; + settingState.disabled = "plan,scout"; await session.refreshMCPTools([dynamicTool]); expect(rebuildCount).toBe(baseline + 1); diff --git a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts index 668222b22..62011854a 100644 --- a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts +++ b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts @@ -10,19 +10,19 @@ function agentByName(agents: AgentDefinition[], name: string): AgentDefinition { } describe("task agent capability descriptions", () => { - it("classifies bundled explore as the only read-only delegated agent", () => { + it("classifies bundled scout as the only read-only delegated agent", () => { const agents = loadBundledAgents(); - expect(isReadOnlyAgent(agentByName(agents, "explore"))).toBe(true); + expect(isReadOnlyAgent(agentByName(agents, "scout"))).toBe(true); for (const name of ["task", "sonic", "plan", "reviewer", "designer"]) { expect(isReadOnlyAgent(agentByName(agents, name))).toBe(false); } }); - it("disables read summarization for explore and librarian, leaves other agents summarizing", () => { + it("disables read summarization for scout and librarian, leaves other agents summarizing", () => { const agents = loadBundledAgents(); - expect(agentByName(agents, "explore").readSummarize).toBe(false); + expect(agentByName(agents, "scout").readSummarize).toBe(false); expect(agentByName(agents, "librarian").readSummarize).toBe(false); for (const name of ["task", "sonic", "plan", "reviewer", "designer"]) { expect(agentByName(agents, name).readSummarize).toBeUndefined(); From 558e0bd085f6f25dc1c5dc20c046dd8816273c4b Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 12:58:54 +0200 Subject: [PATCH 064/205] test(coding-agent): aligned advisor and reload tests with uuidv7 session ids - Updated advisor provider-options parity assertions to the UUIDv7 provider session identity introduced for issue #5040 instead of the retired -advisor suffix. - Disabled codex websocket prewarm in the responses-replay harness so seeded provider-state stubs are not replaced before reload closes them. --- .../advisor-provider-options-parity.test.ts | 24 ++++++++++++------- ...nt-session-openai-responses-replay.test.ts | 7 +++++- 2 files changed, 21 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/test/advisor-provider-options-parity.test.ts b/packages/coding-agent/test/advisor-provider-options-parity.test.ts index 6b83ec0fd..8e54cb18f 100644 --- a/packages/coding-agent/test/advisor-provider-options-parity.test.ts +++ b/packages/coding-agent/test/advisor-provider-options-parity.test.ts @@ -23,6 +23,11 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { TempDir } from "@oh-my-pi/pi-utils"; +/** Provider-facing advisor session ids must be UUIDv7 (issue #5040): Codex writes + * them verbatim onto `conversation_id`/`session_id` headers, so `-advisor` + * labels stay local-only (telemetry, transcripts). */ +const UUID_V7_PATTERN = /^[0-9a-f]{8}-[0-9a-f]{4}-7[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i; + describe("AgentSession advisor provider-options parity", () => { let sharedDir: TempDir; let authStorage: AuthStorage; @@ -99,12 +104,12 @@ describe("AgentSession advisor provider-options parity", () => { // Anthropic fast-mode fallbacks consistent across the two agents. expect(advisor.providerSessionState).toBe(session.providerSessionState); - // Stable advisor-scoped cache key keeps consecutive advisor turns on the - // same OpenAI Responses shard. With no parent `providerPromptCacheKey` - // the main agent's effective key is just its `sessionId`, so the - // advisor's derived key matches its own `${sessionId}-advisor`. - expect(advisor.sessionId).toMatch(/-advisor$/); - expect(advisor.promptCacheKey).toBe(`${mainAgent.sessionId}-advisor`); + // The advisor's session identity is its own provider-facing UUIDv7 + // (issue #5040), distinct from the parent's. Without a pinned parent + // `promptCacheKey` the advisor caches on that same UUID so consecutive + // advisor turns stay on one OpenAI Responses shard. + expect(advisor.sessionId).toMatch(UUID_V7_PATTERN); + expect(advisor.sessionId).not.toBe(mainAgent.sessionId); expect(advisor.promptCacheKey).toBe(advisor.sessionId); }); @@ -197,9 +202,10 @@ describe("AgentSession advisor provider-options parity", () => { // Explicit provider cache keys are shared byte-for-byte with the parent // live turn; only the provider session id stays advisor-scoped. expect(advisor.promptCacheKey).toBe(parentPromptCacheKey); - // Session id is still advisor-scoped so credential stickiness and the - // advisor's session-keyed telemetry stay distinct from the parent. - expect(advisor.sessionId).toMatch(/-advisor$/); + // Session id remains a distinct provider-facing UUIDv7 (issue #5040) so + // credential stickiness and session-keyed telemetry stay distinct from + // the parent. + expect(advisor.sessionId).toMatch(UUID_V7_PATTERN); expect(advisor.sessionId).not.toBe(advisor.promptCacheKey); }); }); diff --git a/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts b/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts index c00be9666..0837c416a 100644 --- a/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts +++ b/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts @@ -258,7 +258,12 @@ async function createSessionHarness( modelRegistry: sharedModelRegistry, sessionManager, model, - settings: Settings.isolated(), + // These tests seed bare `{ close }` stubs into `providerSessionState` and + // assert reload closes them. The SDK's fire-and-forget Codex websocket + // prewarm (models with `preferWebsockets`) would race in and replace the + // stub via `getCodexProviderSessionState`, orphaning the spy — disable + // websockets since these tests exercise reload semantics, not transport. + settings: Settings.isolated({ "providers.openaiWebsockets": "off" }), disableExtensionDiscovery: true, skills: [], contextFiles: [], From 6813fa128fcb0be52850475594092fb593fad34f Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 13:21:06 +0200 Subject: [PATCH 065/205] test: aligned ai and tui test contracts with recent stream and paint changes - Asserted the terminal [DONE] sentinel frame in raw SSE capture since onSseEvent observers receive every wire frame as it arrives. - Included updatedAtMs in credential block persistence expectations per the broker snapshot contract from issue #4980. - Taught VirtualTerminal the no-ED2 destructive paint bytes: ED3 history clears now invalidate the offset-keyed text cache instead of bypassing emulation, keeping legacy full-clear recreation only for the WASM trap workaround. - Refreshed the advisor parity comment for UUIDv7 provider session ids. --- .../auth-storage-block-persistence.test.ts | 11 ++++- packages/ai/test/raw-sse-sdk-capture.test.ts | 8 +++- .../advisor-provider-options-parity.test.ts | 4 +- packages/tui/test/virtual-terminal.ts | 44 +++++++++++++------ 4 files changed, 49 insertions(+), 18 deletions(-) diff --git a/packages/ai/test/auth-storage-block-persistence.test.ts b/packages/ai/test/auth-storage-block-persistence.test.ts index 5c460373b..6488fc10e 100644 --- a/packages/ai/test/auth-storage-block-persistence.test.ts +++ b/packages/ai/test/auth-storage-block-persistence.test.ts @@ -122,8 +122,16 @@ describe("AuthStorage credential block persistence", () => { blockedUntilMs: FUTURE_BLOCK_MS, }); + // `updatedAtMs` is the row's DB write time (issue #4980: same-deadline + // refreshes must be observable), so only its presence is asserted. expect(storage.listCredentialBlocks([row.id])).toEqual([ - { credentialId: row.id, providerKey: PROVIDER_KEY, blockScope: "tier:fable", blockedUntilMs: longerBlock }, + { + credentialId: row.id, + providerKey: PROVIDER_KEY, + blockScope: "tier:fable", + blockedUntilMs: longerBlock, + updatedAtMs: expect.any(Number), + }, ]); } finally { storage.close(); @@ -157,6 +165,7 @@ describe("AuthStorage credential block persistence", () => { providerKey: PROVIDER_KEY, blockScope: "tier:fable", blockedUntilMs: FUTURE_BLOCK_MS, + updatedAtMs: expect.any(Number), }, ]); diff --git a/packages/ai/test/raw-sse-sdk-capture.test.ts b/packages/ai/test/raw-sse-sdk-capture.test.ts index 6399592ae..b0ac1f1fa 100644 --- a/packages/ai/test/raw-sse-sdk-capture.test.ts +++ b/packages/ai/test/raw-sse-sdk-capture.test.ts @@ -226,9 +226,15 @@ describe("SDK raw SSE capture", () => { }).result(); expect(result.stopReason).toBe("stop"); - expect(observed.map(event => event.event)).toEqual(["chat.completion.chunk", "chat.completion.chunk"]); + // Observers receive every raw wire frame as it arrives (types.ts + // `onSseEvent` contract), including the terminal `[DONE]` sentinel, + // which carries no resolvable event name. + expect(observed.map(event => event.event)).toEqual(["chat.completion.chunk", "chat.completion.chunk", null]); expect(JSON.parse(observed[0]!.data)).toEqual(chunks[0]); expect(observed[0]!.raw).toEqual(["event: chat.completion.chunk", `data: ${JSON.stringify(chunks[0])}`]); + const sentinel = observed.at(-1); + expect(sentinel?.data).toBe("[DONE]"); + expect(sentinel?.raw).toEqual(["data: [DONE]"]); }); it("records Azure OpenAI Responses SDK events from the decoded stream", async () => { diff --git a/packages/coding-agent/test/advisor-provider-options-parity.test.ts b/packages/coding-agent/test/advisor-provider-options-parity.test.ts index 8e54cb18f..751388dc9 100644 --- a/packages/coding-agent/test/advisor-provider-options-parity.test.ts +++ b/packages/coding-agent/test/advisor-provider-options-parity.test.ts @@ -167,8 +167,8 @@ describe("AgentSession advisor provider-options parity", () => { expect(opts.onPayload).toBe(onPayload); // Cache routing identity threaded through into the actual stream call. - // Without a parent `providerPromptCacheKey`, advisor's effective key - // collapses to `${main.sessionId}-advisor` which equals its sessionId. + // Without a parent `providerPromptCacheKey`, the advisor's effective key + // is its own provider-facing UUIDv7 session id (issue #5040). expect(opts.sessionId).toBe(advisor.sessionId); expect(opts.promptCacheKey).toBe(advisor.sessionId); expect(opts.providerSessionState).toBe(session.providerSessionState); diff --git a/packages/tui/test/virtual-terminal.ts b/packages/tui/test/virtual-terminal.ts index 64d0b4b18..3821ba0c8 100644 --- a/packages/tui/test/virtual-terminal.ts +++ b/packages/tui/test/virtual-terminal.ts @@ -143,8 +143,9 @@ export class VirtualTerminal implements Terminal { // allocator exhausts after enough cumulative write volume in one instance // (recommit-heavy stress runs hit it); on an OOM trap the wrapper rebuilds // a fresh engine and replays this log, which reproduces the exact terminal - // state. Full-clear recreates reset the log (prior history is erased), so - // it stays bounded by the bytes since the last destructive replay. + // state. Recreates (reset/clear/legacy full-clear) reset the log, so it + // stays bounded by the bytes written since the last fresh engine; the + // no-ED2 destructive paint path leaves it intact. #eventLog: (string | { columns: number; rows: number })[] = []; #eventLogBytes = 0; #logBaseColumns: number; @@ -152,9 +153,10 @@ export class VirtualTerminal implements Terminal { #replayingLog = false; // Memoized text of committed scrollback rows, keyed by absolute offset. Safe // because the engine never evicts (its byte budget sits far above the line - // cap), so an offset's content is stable until a resize (rewrap) or recreate - // (clear) — both reset this. Eliminates the per-op O(history) WASM re-reads - // that made long streaming runs O(n²) in committed rows. + // cap), so an offset's content is stable until a resize (rewrap), a recreate + // (clear), or an ED3 history clear (renumbers offsets) — all reset this. + // Eliminates the per-op O(history) WASM re-reads that made long streaming + // runs O(n²) in committed rows. #historyTextCache: string[] = []; constructor(columns = 80, rows = 24, scrollback?: number) { @@ -429,8 +431,12 @@ export class VirtualTerminal implements Terminal { #engineWrite(data: string): void { const wasBottom = this.#atBottom(); const clearScrollbackAfterFullClear = "\x1b[2J\x1b[H\x1b[3J"; - const clearIndex = data.indexOf(clearScrollbackAfterFullClear); - if (clearIndex >= 0 && this.#canRecreateForFullClear(data, clearIndex)) { + // Destructive full paints emit home + ED3 without ED2 (TUI#emitFullPaint + // rewrites every visible row with self-clearing lines). + const destructiveClear = "\x1b[H\x1b[3J"; + const fullClearIndex = data.indexOf(clearScrollbackAfterFullClear); + const destructiveIndex = data.indexOf(destructiveClear); + if (fullClearIndex >= 0 && this.#clearFollowsPaintBegin(data, fullClearIndex)) { // ghostty-web 0.4 can trap in WASM when libghostty-vt processes a // full-clear + ED3 repaint against an existing history buffer. The // sequence's observable effect here is a blank terminal with empty @@ -438,12 +444,21 @@ export class VirtualTerminal implements Terminal { // state directly in a fresh WASM instance and feed Ghostty the // unmodified text/SGR tail. this.#recreate(); - data = data.slice(0, clearIndex) + data.slice(clearIndex + clearScrollbackAfterFullClear.length); - } else if (this.#pendingEngineResize) { - this.#term.resize(this.#columns, this.#rows); - this.#eventLog.push({ columns: this.#columns, rows: this.#rows }); - this.#historyTextCache.length = 0; // engine rewraps scrollback on resize - this.#pendingEngineResize = false; + data = data.slice(0, fullClearIndex) + data.slice(fullClearIndex + clearScrollbackAfterFullClear.length); + } else { + if (this.#pendingEngineResize) { + this.#term.resize(this.#columns, this.#rows); + this.#eventLog.push({ columns: this.#columns, rows: this.#rows }); + this.#historyTextCache.length = 0; // engine rewraps scrollback on resize + this.#pendingEngineResize = false; + } + if (destructiveIndex >= 0 && this.#clearFollowsPaintBegin(data, destructiveIndex)) { + // ED3 renumbers scrollback offsets, so the offset-keyed history text + // cache is stale. Let Ghostty process the bytes natively — recreating + // to a blank grid here would mask self-clear regressions in the + // no-ED2 repaint contract that the render tests exist to catch. + this.#historyTextCache.length = 0; + } } data = this.#stripSynchronizedOutput(data); data = stripCombiningMarksForGhostty(data); @@ -589,7 +604,8 @@ export class VirtualTerminal implements Terminal { } } - #canRecreateForFullClear(data: string, clearIndex: number): boolean { + /** Whether a viewport/history clear sequence sits immediately after a full-paint begin prefix. */ + #clearFollowsPaintBegin(data: string, clearIndex: number): boolean { const paintBegin = "\x1b[?25l\x1b[?2026h\x1b[?7l"; const paintBeginNoSync = "\x1b[?25l\x1b[?7l"; return ( From d435385ab1515c1f76f552fc4f3a974030787a41 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 13:39:33 +0200 Subject: [PATCH 066/205] feat: introduced max reasoning effort tier across model and rpc systems - Introduced `Max` as a first-class reasoning effort tier across all packages, including AI providers, coding agent configurations, and RPC protocols. - Refactored model effort ladders to use wire-exact mappings and removed legacy effort aliasing (e.g., `max-to-xhigh` mapping). - Updated model registry and provider configurations to support `Max` tier routing, color themes, and UI icon associations. - Expanded test suites to provide end-to-end coverage for the new reasoning tier, including updated compatibility and fallback scenarios. --- docs/models.md | 4 +- docs/rpc.md | 2 +- docs/settings.md | 6 +- packages/agent/CHANGELOG.md | 4 + packages/agent/README.md | 2 +- packages/agent/src/compaction/compaction.ts | 2 + packages/agent/src/thinking.ts | 1 + packages/ai/CHANGELOG.md | 5 + packages/ai/src/providers/amazon-bedrock.ts | 1 + .../providers/anthropic-messages-server.ts | 18 + .../src/providers/azure-openai-responses.ts | 2 +- packages/ai/src/providers/ollama.ts | 2 +- .../providers/openai-chat-server-schema.ts | 2 +- .../ai/src/providers/openai-chat-server.ts | 9 +- packages/ai/src/providers/openai-chat-wire.ts | 2 +- .../src/providers/openai-codex-responses.ts | 2 +- .../openai-codex/request-transformer.ts | 8 +- .../ai/src/providers/openai-completions.ts | 2 +- .../src/providers/openai-responses-server.ts | 9 +- .../ai/src/providers/openai-responses-wire.ts | 16 +- packages/ai/src/providers/openai-responses.ts | 2 +- packages/ai/src/providers/openai-shared.ts | 2 +- packages/ai/src/stream.ts | 7 +- packages/ai/test/anthropic-alignment.test.ts | 69 +- .../auth-gateway-anthropic-messages.test.ts | 27 + .../test/deepseek-reasoning-content.test.ts | 38 +- .../ai/test/glm-5.2-reasoning-effort.test.ts | 15 +- packages/ai/test/max-effort-wire.test.ts | 158 + .../ollama-reasoning-effort-backfill.test.ts | 21 +- packages/ai/test/openai-codex.test.ts | 25 +- .../ai/test/openai-completions-compat.test.ts | 31 +- .../openai-reasoning-effort-fallback.test.ts | 13 +- packages/ai/test/stream.test.ts | 2 +- packages/catalog/CHANGELOG.md | 7 + packages/catalog/src/compat/openai.ts | 33 +- packages/catalog/src/effort.ts | 2 + packages/catalog/src/model-thinking.ts | 277 +- packages/catalog/src/models.json | 2684 ++++------------- .../catalog/src/provider-models/ollama.ts | 3 +- .../src/provider-models/openai-compat.ts | 20 +- packages/catalog/src/variant-collapse.ts | 202 +- .../catalog/test/generated-policies.test.ts | 6 +- .../catalog/test/litellm-provider.test.ts | 6 +- packages/catalog/test/model-thinking.test.ts | 213 +- .../test/ollama-cloud-provider.test.ts | 9 +- packages/catalog/test/ollama-provider.test.ts | 51 +- packages/catalog/test/sakana-provider.test.ts | 8 +- packages/catalog/test/umans-provider.test.ts | 11 +- .../catalog/test/variant-collapse.test.ts | 40 + packages/coding-agent/CHANGELOG.md | 10 +- .../coding-agent/src/config/model-registry.ts | 2 +- .../coding-agent/src/config/model-resolver.ts | 26 +- .../src/config/models-config-schema.ts | 5 +- .../src/config/settings-schema.ts | 5 +- .../modes/theme/defaults/dark-poimandres.json | 11 +- .../theme/defaults/light-poimandres.json | 11 +- .../src/modes/theme/theme-schema.json | 8 +- .../coding-agent/src/modes/theme/theme.ts | 42 +- packages/coding-agent/src/sdk.ts | 4 +- .../coding-agent/src/session/agent-session.ts | 11 +- .../coding-agent/src/system-prompt.test.ts | 31 +- packages/coding-agent/src/thinking.ts | 23 +- .../test/agent-session-retry-fallback.test.ts | 8 +- .../test/agent-session-role-thinking.test.ts | 98 +- .../test/auto-thinking-classifier.test.ts | 30 +- .../test/cli-hide-thinking-flag.test.ts | 4 +- .../coding-agent/test/cli/completions.test.ts | 2 +- .../coding-agent/test/model-resolver.test.ts | 85 +- ...model-selector-role-badge-thinking.test.ts | 22 +- .../test/sdk-model-selection.test.ts | 4 +- packages/terminal-bench/README.md | 2 +- packages/terminal-bench/src/runner.ts | 2 +- .../typescript-edit-benchmark/src/index.ts | 2 +- python/omp-rpc/src/omp_rpc/protocol.py | 8 +- python/omp-rpc/uv.lock | 8 + python/robomp/src/config.py | 2 +- python/robomp/src/pragmas.py | 5 +- scripts/edit_benchmark_common.py | 4 +- 78 files changed, 1688 insertions(+), 2868 deletions(-) create mode 100644 packages/ai/test/max-effort-wire.test.ts create mode 100644 python/omp-rpc/uv.lock diff --git a/docs/models.md b/docs/models.md index d1144a222..5ca430d03 100644 --- a/docs/models.md +++ b/docs/models.md @@ -419,7 +419,7 @@ So a model can exist in registry but not be selectable until auth is available. - exact model id (provider inferred) - fuzzy/substring matching - glob scope patterns in `--models` (e.g. `openai/*`, `*sonnet*`) -- optional `:thinkingLevel` suffix (`off|minimal|low|medium|high|xhigh`) +- optional `:thinkingLevel` suffix (`off|minimal|low|medium|high|xhigh|max`) `--provider` is legacy; `--model` is preferred. @@ -582,7 +582,7 @@ Reasoning / thinking: - `supportsReasoningEffort` — accept `reasoning_effort`. Default: auto (off for Grok, Z.ai/Zhipu, and Xiaomi MiMo). - `supportsReasoningParams` — whether request shaping may send reasoning params at all. Default: auto (off for GitHub Copilot chat-completions). -- `reasoningEffortMap` — partial map from internal effort levels (`minimal|low|medium|high|xhigh`) to provider-specific strings (e.g. DeepSeek maps `xhigh -> "max"`). +- `reasoningEffortMap` — partial map from internal effort levels (`minimal|low|medium|high|xhigh|max`) to provider-specific strings (e.g. Fireworks GLM maps `minimal -> "none"`). - `thinkingFormat` — request shape for thinking: `"openai"` (`reasoning_effort`), `"openrouter"` (`reasoning: { effort }`), `"zai"` (`thinking: { type: "enabled" }`), `"qwen"` (top-level `enable_thinking`), or `"qwen-chat-template"` (`chat_template_kwargs.enable_thinking`). Default: `"openai"`. - `reasoningContentField` — assistant field carrying chain-of-thought: `"reasoning_content"`, `"reasoning"`, or `"reasoning_text"`. Default: auto. - `requiresReasoningContentForToolCalls` — assistant tool-call turns must round-trip the reasoning field (DeepSeek-R1, Kimi, OpenRouter when reasoning is on). Default: `false`. diff --git a/docs/rpc.md b/docs/rpc.md index 6e23cf86b..973b80e1b 100644 --- a/docs/rpc.md +++ b/docs/rpc.md @@ -192,7 +192,7 @@ Local-only slash commands may emit `command_output` frames before completing via ```json { "model": { "provider": "...", "id": "..." }, - "thinkingLevel": "off|minimal|low|medium|high|xhigh", + "thinkingLevel": "off|minimal|low|medium|high|xhigh|max", "isStreaming": false, "isCompacting": false, "steeringMode": "all|one-at-a-time", diff --git a/docs/settings.md b/docs/settings.md index 7961c7a1b..5db977693 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -282,7 +282,7 @@ Every key below is defined in the settings schema; `omp config list` shows the f ### Models -`modelRoles`, `modelTags`, and `cycleOrder` work together to define the models you can switch between. Role values may carry a thinking suffix (`:minimal`, `:low`, `:medium`, `:high`, `:xhigh`). +`modelRoles`, `modelTags`, and `cycleOrder` work together to define the models you can switch between. Role values may carry a thinking suffix (`:minimal`, `:low`, `:medium`, `:high`, `:xhigh`, `:max`). ```yaml modelRoles: @@ -342,17 +342,19 @@ thinkingBudgets: medium: 8192 high: 16384 xhigh: 32768 + max: 32768 ``` | Key | Type | Default | Values | |---|---|---|---| -| `defaultThinkingLevel` | enum | `high` | `minimal`, `low`, `medium`, `high`, `xhigh`, `auto`. Override per run with `--thinking`. | +| `defaultThinkingLevel` | enum | `high` | `minimal`, `low`, `medium`, `high`, `xhigh`, `max`, `auto`. Override per run with `--thinking`. | | `hideThinkingBlock` | boolean | `false` | Hide thinking blocks in output. `--hide-thinking` sets it for the run (display only). | | `thinkingBudgets.minimal` | number | `1024` | Token budget for the `minimal` level. | | `thinkingBudgets.low` | number | `2048` | Token budget for `low`. | | `thinkingBudgets.medium` | number | `8192` | Token budget for `medium`. | | `thinkingBudgets.high` | number | `16384` | Token budget for `high`. | | `thinkingBudgets.xhigh` | number | `32768` | Token budget for `xhigh`. | +| `thinkingBudgets.max` | number | `32768` | Token budget for `max`. | ### Sampling diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 109f6905a..c8702a17d 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `ThinkingLevel.Max` ("max") above `xhigh`, mapping to the catalog `Effort.Max` tier. + ### Fixed - Fixed remote compaction for Codex Responses Lite models (GPT-5.6 family): both the V1 `/responses/compact` request and the V2 `compaction_trigger` stream now apply the lite rewrite (instructions as an input item, no top-level `instructions`/`tools`, `all_turns` reasoning replay on V2) and send the `x-openai-internal-codex-responses-lite` header, matching codex-rs routing compaction through `build_responses_request`. diff --git a/packages/agent/README.md b/packages/agent/README.md index 87c3b2435..148b7182f 100644 --- a/packages/agent/README.md +++ b/packages/agent/README.md @@ -134,7 +134,7 @@ const agent = new Agent({ initialState: { systemPrompt: string[], model: Model, - thinkingLevel: "off" | "minimal" | "low" | "medium" | "high" | "xhigh", + thinkingLevel: "off" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max", tools: AgentTool[], messages: AgentMessage[], }, diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 772f0b232..8acf7ffb8 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -659,6 +659,8 @@ function effortFromThinkingLevel(level: ThinkingLevel): Effort { return Effort.High; case ThinkingLevel.XHigh: return Effort.XHigh; + case ThinkingLevel.Max: + return Effort.Max; case ThinkingLevel.Off: case ThinkingLevel.Inherit: throw new Error(`effortFromThinkingLevel: ${level} must be handled by caller`); diff --git a/packages/agent/src/thinking.ts b/packages/agent/src/thinking.ts index e89c1e834..3dcd6611f 100644 --- a/packages/agent/src/thinking.ts +++ b/packages/agent/src/thinking.ts @@ -13,6 +13,7 @@ export const ThinkingLevel = { Medium: Effort.Medium, High: Effort.High, XHigh: Effort.XHigh, + Max: Effort.Max, } as const; export type ThinkingLevel = (typeof ThinkingLevel)[keyof typeof ThinkingLevel]; diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7f0c5dc5f..b2c3669e1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,10 +4,14 @@ ### Added +- Added `max` as a first-class reasoning effort option across providers and wire schemas +- Updated `max` reasoning budget to 32768 tokens across all provider budget tables + - Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. - Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`. - Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress. - Added Novita API-key login with authenticated key validation and `NOVITA_API_KEY` discovery ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). +- Added `"max"` as a first-class reasoning effort across provider options, wire types (`reasoning_effort`/`reasoning.effort`), server intake guards, and the Codex request transformer; user `max` now serializes 1:1 to the provider `max` tier instead of being reachable only through the retired shifted effort maps. Inbound Anthropic gateway requests now map `output_config.effort` onto `options.reasoning`. ### Changed @@ -16,6 +20,7 @@ - Standardized Responses Lite activation via model-level catalog flags - Recognized Pro Lite as a paid plan tier for OpenAI Codex models - Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape. +- Changed effort budget tables (`ANTHROPIC_THINKING`, `GOOGLE_THINKING`, `BEDROCK_CLAUDE_THINKING`, Bedrock `defaultBudgets`) to carry a `max` row (32768), and `getGoogleBudget` to resolve `max` to the largest bucket explicitly. ### Fixed diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index f361692bc..b30d8a9aa 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -997,6 +997,7 @@ function buildAdditionalModelRequestFields( medium: 8192, high: 16384, xhigh: 32768, + max: 32768, }; const budget = options.thinkingBudgets?.[level] ?? defaultBudgets[level]; diff --git a/packages/ai/src/providers/anthropic-messages-server.ts b/packages/ai/src/providers/anthropic-messages-server.ts index 480d00931..63616473a 100644 --- a/packages/ai/src/providers/anthropic-messages-server.ts +++ b/packages/ai/src/providers/anthropic-messages-server.ts @@ -1,3 +1,4 @@ +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { logger } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import { captureRequestHeaders, resolvePromptCacheKey } from "../auth-gateway/http"; @@ -291,6 +292,19 @@ function deriveCacheRetention(data: { return strongest; } +/** + * Inbound `output_config.effort` wire literal → catalog `Effort` (1:1). + * Values outside this table (none exist in the schema today) are ignored + * rather than guessed at. + */ +const REASONING_EFFORT_BY_WIRE: Partial> = { + low: Effort.Low, + medium: Effort.Medium, + high: Effort.High, + xhigh: Effort.XHigh, + max: Effort.Max, +}; + export function parseRequest(body: unknown, headers?: Headers): ParsedRequest { const data = anthropicMessagesRequestSchema(body); if (data instanceof type.errors) { @@ -351,6 +365,10 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest { if (data.output_config?.task_budget) { options.taskBudget = data.output_config.task_budget; } + if (data.output_config?.effort) { + const mapped = REASONING_EFFORT_BY_WIRE[data.output_config.effort]; + if (mapped !== undefined) options.reasoning = mapped; + } const cacheRetention = deriveCacheRetention(data); if (cacheRetention !== undefined) options.cacheRetention = cacheRetention; // Anthropic clients commonly send `metadata: { user_id }`; forward verbatim diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 31090702c..2c8e73430 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -56,7 +56,7 @@ function resolveDeploymentName(model: Model<"azure-openai-responses">, options?: // Azure OpenAI Responses-specific options export interface AzureOpenAIResponsesOptions extends StreamOptions { - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "detailed" | "concise" | null; azureApiVersion?: string; azureResourceName?: string; diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index a0fb862a4..bfcdb65e0 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -35,7 +35,7 @@ import { transformMessages } from "./transform-messages"; import { joinTextWithImagePlaceholder, partitionVisionContent } from "./vision-guard"; export interface OllamaChatOptions extends StreamOptions { - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; disableReasoning?: boolean; toolChoice?: ToolChoice; } diff --git a/packages/ai/src/providers/openai-chat-server-schema.ts b/packages/ai/src/providers/openai-chat-server-schema.ts index bdb1835e8..2490eec53 100644 --- a/packages/ai/src/providers/openai-chat-server-schema.ts +++ b/packages/ai/src/providers/openai-chat-server-schema.ts @@ -209,7 +209,7 @@ export const openaiChatRequestSchema = type({ "frequency_penalty?": "number", "logit_bias?": type({ "[string]": "number" }), "user?": "string", - "reasoning_effort?": "'minimal' | 'low' | 'medium' | 'high' | 'xhigh'", + "reasoning_effort?": "'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max'", "parallel_tool_calls?": "boolean", "service_tier?": "'auto' | 'default' | 'flex' | 'scale' | 'priority'", "metadata?": type({ "[string]": "unknown" }), diff --git a/packages/ai/src/providers/openai-chat-server.ts b/packages/ai/src/providers/openai-chat-server.ts index 441138ea9..d2f8e0bbc 100644 --- a/packages/ai/src/providers/openai-chat-server.ts +++ b/packages/ai/src/providers/openai-chat-server.ts @@ -35,7 +35,14 @@ export type { ParsedRequest }; type ReasoningEffort = NonNullable; function isReasoningEffort(value: unknown): value is ReasoningEffort { - return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh"; + return ( + value === "minimal" || + value === "low" || + value === "medium" || + value === "high" || + value === "xhigh" || + value === "max" + ); } function isServiceTier(value: unknown): value is ServiceTier { diff --git a/packages/ai/src/providers/openai-chat-wire.ts b/packages/ai/src/providers/openai-chat-wire.ts index b2e5f22a0..0ff7f4a9c 100644 --- a/packages/ai/src/providers/openai-chat-wire.ts +++ b/packages/ai/src/providers/openai-chat-wire.ts @@ -116,7 +116,7 @@ export type Metadata = { }; /** Constrains effort on reasoning for reasoning models. */ -export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null; +export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" | null; /** JSON object response format (older JSON mode). */ export interface ResponseFormatJSONObject { diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 609a570fb..a96d5b2a9 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -110,7 +110,7 @@ import { import { transformMessages } from "./transform-messages"; export interface OpenAICodexResponsesOptions extends StreamOptions { - reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "concise" | "detailed" | null; /** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index 8c095137f..a25d5b665 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -8,7 +8,7 @@ import { mapOpenAIReasoningEffort } from "../openai-shared"; export type CodexReasoningContext = "auto" | "current_turn" | "all_turns"; /** User-facing effort levels accepted by Codex request options. */ -type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh"; +type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; /** Caller literal → catalog `Effort` bridge (the enum is nominal). */ const EFFORT_BY_NAME: Record = { @@ -17,6 +17,7 @@ const EFFORT_BY_NAME: Record = { medium: Effort.Medium, high: Effort.High, xhigh: Effort.XHigh, + max: Effort.Max, }; export interface ReasoningConfig { @@ -28,7 +29,7 @@ export interface ReasoningConfig { } export interface CodexRequestOptions { - /** User-facing effort; the wire-only `max` tier is reached via the model's effort map. */ + /** User-facing effort; maps 1:1 onto the wire tier of the same name. */ reasoningEffort?: CodexCallerEffort | "none"; reasoningSummary?: ReasoningConfig["summary"] | null; /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ @@ -99,7 +100,8 @@ export function resolveCodexResponsesLite( /** * Clamp a user-facing effort to the model's ladder, then remap to the wire - * tier (e.g. GPT-5.6's shifted five-tier scale sends `max` for user `xhigh`). + * tier. User efforts map 1:1 onto wire tiers; the effort map only covers + * host quirks where a wire tier genuinely does not exist (e.g. `minimal→none`). * A mapped value outside the Codex wire vocabulary is a broken compat/model * effort map — fail loudly rather than silently sending a different tier. */ diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index f200c268c..83cc11666 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -459,7 +459,7 @@ export function isOpenAICompletionsProgressChunk(chunk: unknown): boolean { export interface OpenAICompletionsOptions extends StreamOptions { toolChoice?: ToolChoice; - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; /** Force-disable reasoning where supported, or request the lowest effort on generic effort endpoints. */ disableReasoning?: boolean; serviceTier?: ServiceTier; diff --git a/packages/ai/src/providers/openai-responses-server.ts b/packages/ai/src/providers/openai-responses-server.ts index 6a4d947d8..e2dd73389 100644 --- a/packages/ai/src/providers/openai-responses-server.ts +++ b/packages/ai/src/providers/openai-responses-server.ts @@ -40,7 +40,14 @@ export type { ParsedRequest }; // ─── narrow guards ────────────────────────────────────────────────────────── function isReasoningEffort(value: unknown): value is NonNullable { - return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh"; + return ( + value === "minimal" || + value === "low" || + value === "medium" || + value === "high" || + value === "xhigh" || + value === "max" + ); } function isServiceTier(value: unknown): value is NonNullable { diff --git a/packages/ai/src/providers/openai-responses-wire.ts b/packages/ai/src/providers/openai-responses-wire.ts index 7992ec319..e194a4c3f 100644 --- a/packages/ai/src/providers/openai-responses-wire.ts +++ b/packages/ai/src/providers/openai-responses-wire.ts @@ -6305,9 +6305,9 @@ export interface Reasoning { /** * Constrains effort on reasoning for * [reasoning models](https://platform.openai.com/docs/guides/reasoning). Currently - * supported values are `none`, `minimal`, `low`, `medium`, `high`, and `xhigh`. - * Reducing reasoning effort can result in faster responses and fewer tokens used - * on reasoning in a response. + * supported values are `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, and + * `max`. Reducing reasoning effort can result in faster responses and fewer + * tokens used on reasoning in a response. * * - `gpt-5.1` defaults to `none`, which does not perform reasoning. The supported * reasoning values for `gpt-5.1` are `none`, `low`, `medium`, and `high`. Tool @@ -6316,6 +6316,7 @@ export interface Reasoning { * support `none`. * - The `gpt-5-pro` model defaults to (and only supports) `high` reasoning effort. * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. + * - `max` is supported for `gpt-5.6` and later models. */ effort?: ReasoningEffort | null; /** @@ -6346,9 +6347,9 @@ export interface Reasoning { /** * Constrains effort on reasoning for * [reasoning models](https://platform.openai.com/docs/guides/reasoning). Currently - * supported values are `none`, `minimal`, `low`, `medium`, `high`, and `xhigh`. - * Reducing reasoning effort can result in faster responses and fewer tokens used - * on reasoning in a response. + * supported values are `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, and + * `max`. Reducing reasoning effort can result in faster responses and fewer tokens + * used on reasoning in a response. * * - `gpt-5.1` defaults to `none`, which does not perform reasoning. The supported * reasoning values for `gpt-5.1` are `none`, `low`, `medium`, and `high`. Tool @@ -6357,8 +6358,9 @@ export interface Reasoning { * support `none`. * - The `gpt-5-pro` model defaults to (and only supports) `high` reasoning effort. * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. + * - `max` is supported for `gpt-5.6` and later models. */ -export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null; +export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" | null; /** * JSON object response format. An older method of generating JSON responses. Using * `json_schema` is recommended for models that support it. Note that the model diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 937cbe87a..2be20dd81 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -95,7 +95,7 @@ import { // OpenAI Responses-specific options export interface OpenAIResponsesOptions extends StreamOptions { - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "detailed" | "concise" | null; serviceTier?: ServiceTier; textVerbosity?: "low" | "medium" | "high"; diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 789f31213..47a803349 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -643,7 +643,7 @@ export type OpenAICompletionsParams = Omit = { medium: 8192, high: 16384, xhigh: 32768, + max: 32768, }; const GOOGLE_THINKING: Record = { @@ -1252,6 +1253,7 @@ const GOOGLE_THINKING: Record = { medium: 8192, high: 16384, xhigh: 24575, + max: 32768, }; const BEDROCK_CLAUDE_THINKING: Record = { @@ -1260,6 +1262,7 @@ const BEDROCK_CLAUDE_THINKING: Record = { medium: 8192, high: 16384, xhigh: 16384, + max: 32768, }; function resolveBedrockThinkingBudget( @@ -1842,7 +1845,9 @@ function getGoogleBudget( return 2048; case "medium": return 8192; - default: + case "high": + case "xhigh": + case "max": return model.id.includes("2.5-flash") ? 24576 : 32768; } } diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index d3e7b42f5..837bf7b21 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -1939,16 +1939,17 @@ describe("Anthropic request fingerprint alignment", () => { }); it("drops sampling params and keeps summarized adaptive thinking for OAuth Opus 4.7+", async () => { + const opus47 = buildModel({ + ...ANTHROPIC_MODEL_SPEC, + id: "claude-opus-4-7", + name: "Claude Opus 4.7", + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + }, + }); const payload = (await captureAnthropicPayload( - buildModel({ - ...ANTHROPIC_MODEL_SPEC, - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - thinking: { - mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - }, - }), + opus47, { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1976,18 +1977,10 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.context_management).toEqual({ edits: [{ type: "clear_thinking_20251015", keep: "all" }], }); - expect(payload.output_config).toEqual({ effort: "xhigh" }); + expect(payload.output_config).toEqual({ effort: "high" }); - const maxPayload = (await captureAnthropicPayload( - buildModel({ - ...ANTHROPIC_MODEL_SPEC, - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - thinking: { - mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - }, - }), + const xhighPayload = (await captureAnthropicPayload( + opus47, { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -2000,6 +1993,23 @@ describe("Anthropic request fingerprint alignment", () => { thinking?: { type?: string; display?: string }; output_config?: { effort?: string }; }; + expect(xhighPayload.thinking).toEqual({ type: "adaptive", display: "summarized" }); + expect(xhighPayload.output_config).toEqual({ effort: "xhigh" }); + + const maxPayload = (await captureAnthropicPayload( + opus47, + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], + }, + { + thinkingEnabled: true, + reasoning: Effort.Max, + }, + )) as { + thinking?: { type?: string; display?: string }; + output_config?: { effort?: string }; + }; expect(maxPayload.thinking).toEqual({ type: "adaptive", display: "summarized" }); expect(maxPayload.output_config).toEqual({ effort: "max" }); }); @@ -2014,15 +2024,14 @@ describe("Anthropic request fingerprint alignment", () => { baseUrl: "https://api.code.umans.ai", thinking: { mode: "anthropic-budget-effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.High, Effort.Max], }, }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], }, - Effort.XHigh, + Effort.Max, )) as { thinking?: { type?: string; budget_tokens?: number }; output_config?: { effort?: string }; @@ -2098,7 +2107,7 @@ describe("Anthropic request fingerprint alignment", () => { name: "Claude Opus 4.7", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2120,7 +2129,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.context_management).toEqual({ edits: [{ type: "clear_thinking_20251015", keep: "all" }], }); - expect(payload.output_config).toEqual({ effort: "xhigh" }); + expect(payload.output_config).toEqual({ effort: "high" }); }); it("sends task budgets through Anthropic output_config without dropping adaptive effort", async () => { @@ -2131,7 +2140,7 @@ describe("Anthropic request fingerprint alignment", () => { name: "Claude Opus 4.7", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2151,7 +2160,7 @@ describe("Anthropic request fingerprint alignment", () => { }; expect(payload.output_config).toEqual({ - effort: "xhigh", + effort: "high", task_budget: { type: "tokens", total: 64_000, remaining: 48_000 }, }); }); @@ -2164,7 +2173,7 @@ describe("Anthropic request fingerprint alignment", () => { name: "Claude Opus 4.7", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2209,7 +2218,7 @@ describe("Anthropic request fingerprint alignment", () => { maxTokens: 128_000, thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2236,7 +2245,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.tool_choice).toEqual({ type: "auto" }); expect(payload.thinking).toEqual({ type: "adaptive", display: "summarized" }); - expect(payload.output_config).toEqual({ effort: "xhigh" }); + expect(payload.output_config).toEqual({ effort: "high" }); } }); diff --git a/packages/ai/test/auth-gateway-anthropic-messages.test.ts b/packages/ai/test/auth-gateway-anthropic-messages.test.ts index dce464548..1df22f39e 100644 --- a/packages/ai/test/auth-gateway-anthropic-messages.test.ts +++ b/packages/ai/test/auth-gateway-anthropic-messages.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { encodeResponse, encodeStream, parseRequest } from "@oh-my-pi/pi-ai/providers/anthropic-messages-server"; import type { AssistantMessage, AssistantMessageEvent, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; function emptyUsage(): AssistantMessage["usage"] { return { @@ -220,6 +221,32 @@ describe("anthropic-messages parseRequest", () => { expect(parsed.context.messages[1]!.role).toBe("toolResult"); }); + it("maps inbound output_config.effort onto options.reasoning 1:1", () => { + const cases = [ + ["low", Effort.Low], + ["medium", Effort.Medium], + ["high", Effort.High], + ["xhigh", Effort.XHigh], + ["max", Effort.Max], + ] as const; + for (const [wire, effort] of cases) { + const parsed = parseRequest({ + model: "m", + max_tokens: 8, + output_config: { effort: wire }, + messages: [{ role: "user", content: "hi" }], + }); + expect(parsed.options.reasoning).toBe(effort); + } + + const absent = parseRequest({ + model: "m", + max_tokens: 8, + messages: [{ role: "user", content: "hi" }], + }); + expect(absent.options.reasoning).toBeUndefined(); + }); + it("rejects missing required fields and unsupported request controls", () => { expect(() => parseRequest({})).toThrow(/model/); expect(() => parseRequest({ model: "m", messages: [] })).toThrow(/max_tokens/); diff --git a/packages/ai/test/deepseek-reasoning-content.test.ts b/packages/ai/test/deepseek-reasoning-content.test.ts index 3cbdfd349..f651f5714 100644 --- a/packages/ai/test/deepseek-reasoning-content.test.ts +++ b/packages/ai/test/deepseek-reasoning-content.test.ts @@ -3,6 +3,7 @@ import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; interface OpenAICompletionAssistantWireMessage { @@ -66,52 +67,37 @@ function assistantToolCall( describe("DeepSeek reasoning_content tool-call replay", () => { // ---------------------------------------------------------------- - // Fix 1: effortMap for DeepSeek-family on any provider + // Fix 1: honest [high, max] ladder for DeepSeek-family on any provider // ---------------------------------------------------------------- - describe("thinking effortMap (Fix 1)", () => { - it("maps unsupported lower DeepSeek efforts to high on opencode-go", () => { + describe("thinking ladder (Fix 1)", () => { + it("bakes the honest [high, max] ladder with no effortMap on opencode-go", () => { const model = deepseekModel({ provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); - it("maps unsupported lower DeepSeek efforts to high on NVIDIA", () => { + it("bakes the honest [high, max] ladder with no effortMap on NVIDIA", () => { const model = deepseekModel({ provider: "nvidia", baseUrl: "https://integrate.api.nvidia.com/v1", id: "deepseek-ai/deepseek-v4-flash", }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); - it("maps unsupported lower DeepSeek efforts to high on the official endpoint", () => { + it("bakes the honest [high, max] ladder with no effortMap on the official endpoint", () => { const model = deepseekModel({ provider: "deepseek", baseUrl: "https://api.deepseek.com/v1", id: "deepseek-v4-pro", }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); it("does NOT map xhigh for non-DeepSeek models", () => { diff --git a/packages/ai/test/glm-5.2-reasoning-effort.test.ts b/packages/ai/test/glm-5.2-reasoning-effort.test.ts index 969e4291f..1cb4246f0 100644 --- a/packages/ai/test/glm-5.2-reasoning-effort.test.ts +++ b/packages/ai/test/glm-5.2-reasoning-effort.test.ts @@ -6,11 +6,12 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; // GLM-5.2 reasoning-effort dialects diverge per host (verified against live -// endpoints): a direct GLM host (Fireworks) wants the top UI tier on the wire -// as `max` while keeping its distinct lower tiers, whereas OpenRouter rejects -// `max` (HTTP 400) and treats `xhigh` as its own max tier. The catalog bakes -// the right `thinking.effortMap`; these tests pin the resulting wire value so a -// future map change can't silently 400 either host. +// endpoints): a direct GLM host (Fireworks) exposes a real `max` top tier and +// keeps its distinct lower tiers (with the `minimal -> none` host quirk), +// whereas OpenRouter rejects `max` (HTTP 400) and treats `xhigh` as its own +// max tier. The catalog bakes the right ladder/`thinking.effortMap`; these +// tests pin the resulting wire value so a future change can't silently 400 +// either host. const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 0 }] }; function chatSse(): Response { @@ -105,8 +106,8 @@ const openRouter = buildModel({ describe("GLM-5.2 reasoning effort wire mapping", () => { afterEach(() => vi.restoreAllMocks()); - it("maps the top tier to reasoning_effort:max on a direct GLM host (Fireworks), lower tiers literal", async () => { - expect(await captureChatEffort(fireworks, Effort.XHigh)).toBe("max"); + it("sends reasoning_effort:max for the real max tier on a direct GLM host (Fireworks), lower tiers literal", async () => { + expect(await captureChatEffort(fireworks, Effort.Max)).toBe("max"); expect(await captureChatEffort(fireworks, Effort.High)).toBe("high"); expect(await captureChatEffort(fireworks, Effort.Medium)).toBe("medium"); // Fireworks rejects literal `minimal`; the host quirk merge keeps `minimal -> none`. diff --git a/packages/ai/test/max-effort-wire.test.ts b/packages/ai/test/max-effort-wire.test.ts new file mode 100644 index 000000000..d716fe221 --- /dev/null +++ b/packages/ai/test/max-effort-wire.test.ts @@ -0,0 +1,158 @@ +import { describe, expect, it, vi } from "bun:test"; +import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; +import { transformRequestBody } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; +import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { createCodexModel } from "./helpers"; + +// End-to-end guard for the first-class `max` reasoning tier: a user-requested +// `reasoning: "max"` on a model whose ladder natively includes `Effort.Max` +// must reach every wire surface verbatim — no aliasing, no clamping. Fixtures +// use explicit thinking ladders and neutral ids so catalog detection cannot +// interfere. + +const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 0 }] }; + +const MAX_LADDER = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max] as const; + +function chatSse(): Response { + const chunk = (delta: unknown, finish: string | null) => + JSON.stringify({ + id: "x", + object: "chat.completion.chunk", + created: 0, + choices: [{ index: 0, delta, finish_reason: finish }], + }); + return new Response(`data: ${chunk({ content: "ok" }, null)}\n\ndata: ${chunk({}, "stop")}\n\ndata: [DONE]\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +function responsesSse(): Response { + return new Response( + `data: ${JSON.stringify({ + type: "response.completed", + response: { + status: "completed", + usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2, input_tokens_details: { cached_tokens: 0 } }, + }, + })}\n\n`, + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); +} + +describe("first-class max reasoning tier wire coverage", () => { + it("sends reasoning_effort:max on Chat Completions", async () => { + const model: Model<"openai-completions"> = buildModel({ + id: "max-wire-chat", + name: "Max Wire Chat", + api: "openai-completions", + provider: "custom", + baseUrl: "https://chat.example.test/v1", + reasoning: true, + compat: { + thinkingFormat: "openai", + supportsReasoningParams: true, + supportsReasoningEffort: true, + }, + thinking: { mode: "effort", efforts: MAX_LADDER }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_384, + }); + + let body: Record | undefined; + const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + body = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; + return chatSse(); + }); + for await (const event of streamOpenAICompletions(model, context, { + apiKey: "k", + fetch: fetchMock, + reasoning: "max", + })) { + if (event.type === "done" || event.type === "error") break; + } + expect(body?.reasoning_effort).toBe("max"); + }); + + it("sends reasoning.effort:max on the Responses surface", async () => { + const model: Model<"openai-responses"> = buildModel({ + id: "max-wire-responses", + name: "Max Wire Responses", + api: "openai-responses", + provider: "custom-responses", + baseUrl: "https://responses.example.test/v1", + reasoning: true, + compat: { + supportsReasoningParams: true, + supportsReasoningEffort: true, + }, + thinking: { mode: "effort", efforts: MAX_LADDER }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_384, + }); + + let body: Record | undefined; + const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + body = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; + return responsesSse(); + }); + for await (const event of streamOpenAIResponses(model, context, { + apiKey: "k", + fetch: fetchMock, + reasoning: "max", + })) { + if (event.type === "done" || event.type === "error") break; + } + const reasoningParam = body?.reasoning as { effort?: string } | undefined; + expect(reasoningParam?.effort).toBe("max"); + }); + + it("sends reasoning.effort:max through the Codex request transformer", async () => { + const model = createCodexModel("codex-max-wire", { + thinking: { mode: "effort", efforts: MAX_LADDER }, + }); + const transformed = await transformRequestBody({ model: model.id, input: [] }, model, { + reasoningEffort: "max", + }); + expect(transformed.reasoning?.effort).toBe("max"); + }); + + it("sends output_config.effort:max on Anthropic adaptive thinking", async () => { + const model = buildModel({ + id: "adaptive-max-wire", + name: "Adaptive Max Wire", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + thinking: { mode: "anthropic-adaptive", efforts: MAX_LADDER }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }); + + const { promise, resolve } = Promise.withResolvers(); + const controller = new AbortController(); + controller.abort(); + streamAnthropic(model, context, { + apiKey: "sk-ant-test", + isOAuth: false, + signal: controller.signal, + thinkingEnabled: true, + reasoning: Effort.Max, + onPayload: payload => resolve(payload), + }); + const payload = (await promise) as { output_config?: { effort?: string } }; + expect(payload.output_config).toEqual({ effort: "max" }); + }); +}); diff --git a/packages/ai/test/ollama-reasoning-effort-backfill.test.ts b/packages/ai/test/ollama-reasoning-effort-backfill.test.ts index 8cafe1f2a..04d0ea2af 100644 --- a/packages/ai/test/ollama-reasoning-effort-backfill.test.ts +++ b/packages/ai/test/ollama-reasoning-effort-backfill.test.ts @@ -3,6 +3,7 @@ import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-response import type { Context } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { clampThinkingLevelForModel, getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; const testContext: Context = { messages: [{ role: "user", content: "hi", timestamp: 0 }], @@ -14,11 +15,13 @@ function abortedSignal(): AbortSignal { return controller.signal; } -describe("ollama reasoning effort backfill reaches the Responses wire", () => { - it("sends low instead of minimal for a stale ollama spec carrying no effort map", async () => { - // Reproduces the HTTP 400 `invalid reasoning value: "minimal"` path: a - // reasoning-capable Ollama model whose cached/custom spec predates the - // remap. buildModel must backfill the effort map so the wire sends `low`. +describe("ollama effort ladder normalization reaches the Responses wire", () => { + it("normalizes a stale ollama spec and sends native max on the wire", async () => { + // A cached/custom spec from before the wire-exact ladder existed: + // reasoning-capable with `minimal` offered. buildModel must normalize + // the ladder to Ollama's low/medium/high/max vocabulary so requests at + // the top tier serialize `max` verbatim (HTTP 400 `invalid reasoning + // value: "minimal"` was the historical failure of the stale surface). const model = buildModel({ id: "gemma4:e4b", name: "gemma4:e4b", @@ -33,16 +36,20 @@ describe("ollama reasoning effort backfill reaches the Responses wire", () => { thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, }); + // The stale `minimal` tier is gone; selecting it clamps to the floor. + expect(getSupportedEfforts(model)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(clampThinkingLevelForModel(model, Effort.Minimal)).toBe(Effort.Low); + const { promise, resolve } = Promise.withResolvers>(); streamOpenAIResponses(model, testContext, { apiKey: "test-key", signal: abortedSignal(), - reasoning: "minimal", + reasoning: "max", reasoningSummary: "auto", onPayload: payload => resolve(payload as Record), }); const payload = await promise; - expect(payload.reasoning).toEqual({ effort: "low", summary: "auto" }); + expect(payload.reasoning).toEqual({ effort: "max", summary: "auto" }); }); }); diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index 1c7492afc..d174d7071 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -312,28 +312,29 @@ describe("openai-codex reasoning effort validation", () => { transformRequestBody({ ...body }, createCodexModel(body.model), { reasoningEffort: "xhigh" }), ).rejects.toThrow(/Supported efforts: medium, high/); }); + + it("rejects gpt-5.6 minimal now that the wire floor is low", async () => { + const body: RequestBody = { model: "gpt-5.6-sol", input: [] }; + await expect( + transformRequestBody(body, createCodexModel(body.model), { reasoningEffort: "minimal" }), + ).rejects.toThrow(/Supported efforts: low, medium, high, xhigh, max/); + }); }); describe("openai-codex reasoning effort wire mapping", () => { - it("shifts gpt-5.6 user efforts one wire tier up via the baked effort map", async () => { + it("maps gpt-5.6 user efforts 1:1 onto wire tiers", async () => { const model = createCodexModel("gpt-5.6-sol"); - const shifted = [ - ["minimal", "low"], - ["low", "medium"], - ["medium", "high"], - ["high", "xhigh"], - ["xhigh", "max"], - ] as const; + const efforts = ["low", "medium", "high", "xhigh", "max"] as const; - for (const [requested, wire] of shifted) { + for (const effort of efforts) { const transformed = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: requested, + reasoningEffort: effort, }); - expect(transformed.reasoning?.effort).toBe(wire); + expect(transformed.reasoning?.effort).toBe(effort); } }); - it("keeps pre-5.6 efforts unshifted and passes none through unmapped", async () => { + it("keeps pre-5.6 efforts 1:1 and passes none through unmapped", async () => { const gpt55 = createCodexModel("gpt-5.5"); const unshifted = await transformRequestBody({ model: gpt55.id }, gpt55, { reasoningEffort: "xhigh" }); expect(unshifted.reasoning?.effort).toBe("xhigh"); diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 728124025..4c9d8ff0e 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -115,7 +115,7 @@ function kimiZaiModel(): Model<"openai-completions"> { async function captureOpenAICompletionsPayload( model: Model<"openai-completions">, context: Context = baseContext(), - options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" }, + options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max" }, ): Promise { const { promise, resolve } = Promise.withResolvers(); const fetchMock = createMockFetch(["[DONE]"]); @@ -810,7 +810,7 @@ describe("openai-completions compatibility", () => { expect(getNestedBoolean(chatTemplateArgs, "enable_thinking")).toBe(true); }); - it("maps GLM-5.2 xhigh to Z.AI max and enables tool streaming", async () => { + it("sends reasoning_effort:max for the real Z.AI max tier and enables tool streaming", async () => { const model = zaiGlm52Model(); const readTool: Tool = { name: "read", @@ -828,7 +828,7 @@ describe("openai-completions compatibility", () => { { ...baseContext(), tools: [readTool] }, { apiKey: "test-key", - reasoning: "xhigh", + reasoning: "max", signal: createAbortedSignal(), onPayload: payload => resolve(payload), maxTokens: 65_536, @@ -879,21 +879,10 @@ describe("openai-completions compatibility", () => { expect(payloadObject?.tool_stream).toBeUndefined(); }); - it("maps GLM-5.2 minimal reasoning to disabled Z.AI thinking", async () => { + it("bakes the honest [high, max] Z.AI GLM-5.2 ladder with no effortMap", () => { const model = zaiGlm52Model(); - - const { promise, resolve } = Promise.withResolvers(); - streamOpenAICompletions(model, baseContext(), { - apiKey: "test-key", - reasoning: "minimal", - signal: createAbortedSignal(), - onPayload: payload => resolve(payload), - }); - const payload = await promise; - const thinking = getNestedObject(payload, "thinking"); - - expect(thinking?.type).toBe("disabled"); - expect(toObject(payload)?.reasoning_effort).toBeUndefined(); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); it("treats finish_reason end as stop", async () => { @@ -1217,7 +1206,7 @@ describe("kimi model detection via detectCompat", () => { expect(openRouterKimi.compat.thinkingFormat).toBe("openrouter"); }); - it("maps OpenRouter Anthropic adaptive reasoning efforts to the Anthropic scale", async () => { + it("sends OpenRouter Anthropic adaptive reasoning efforts 1:1 on the wire", async () => { const model: Model<"openai-completions"> = buildModel({ ...gpt4oMiniSpec, api: "openai-completions", @@ -1229,9 +1218,11 @@ describe("kimi model detection via detectCompat", () => { const highPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "high" }); const xhighPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "xhigh" }); + const maxPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "max" }); - expect(getNestedObject(highPayload, "reasoning")).toEqual({ effort: "xhigh" }); - expect(getNestedObject(xhighPayload, "reasoning")).toEqual({ effort: "max" }); + expect(getNestedObject(highPayload, "reasoning")).toEqual({ effort: "high" }); + expect(getNestedObject(xhighPayload, "reasoning")).toEqual({ effort: "xhigh" }); + expect(getNestedObject(maxPayload, "reasoning")).toEqual({ effort: "max" }); }); // Regression for #1071: OpenCode-Go/Zen handle reasoning content server-side diff --git a/packages/ai/test/openai-reasoning-effort-fallback.test.ts b/packages/ai/test/openai-reasoning-effort-fallback.test.ts index 1676e29ff..991be7df8 100644 --- a/packages/ai/test/openai-reasoning-effort-fallback.test.ts +++ b/packages/ai/test/openai-reasoning-effort-fallback.test.ts @@ -171,10 +171,10 @@ function createResponsesModel(): Model<"openai-responses"> { maxTokens: 16_384, }); } -function createMappedResponsesModel(): Model<"openai-responses"> { +function createMaxLadderResponsesModel(): Model<"openai-responses"> { return buildModel({ - id: "mapped-responses-reasoner", - name: "Mapped Responses Reasoner", + id: "max-ladder-responses-reasoner", + name: "Max Ladder Responses Reasoner", api: "openai-responses", provider: "custom-responses", baseUrl: "https://responses.example.test/v1", @@ -185,8 +185,7 @@ function createMappedResponsesModel(): Model<"openai-responses"> { }, thinking: { mode: "effort", - efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, input: ["text"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, @@ -285,10 +284,10 @@ describe("OpenAI reasoning effort fallback retry", () => { { preconnect: fetch.preconnect }, ); - const result = await streamOpenAIResponses(createMappedResponsesModel(), testContext, { + const result = await streamOpenAIResponses(createMaxLadderResponsesModel(), testContext, { apiKey: "test-key", fetch: fetchMock, - reasoning: "xhigh", + reasoning: "max", }).result(); expect(result.stopReason).toBe("stop"); diff --git a/packages/ai/test/stream.test.ts b/packages/ai/test/stream.test.ts index ce1769430..c48f90d83 100644 --- a/packages/ai/test/stream.test.ts +++ b/packages/ai/test/stream.test.ts @@ -1765,7 +1765,7 @@ describe("Generate E2E Tests", () => { tools: [calculatorTool], }, { - reasoning: Effort.XHigh, + reasoning: Effort.Max, interleavedThinking: true, onPayload: payload => { capturedPayload = payload; diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 1df563b14..02e7eee1e 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -10,10 +10,17 @@ - Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) - Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). - Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. +- Added `Effort.Max` ("max") as a first-class user-facing thinking level above `xhigh`. ### Changed +- Standardized reasoning effort levels to use a wire-exact `max` tier across all model providers +- Refactored Devin model routing to support 1:1 mapping for the `max` effort tier +- Normalized stale Ollama model configurations to the new wire-exact effort ladder + - Updated costs and context windows for various models in the catalog +- **Breaking**: Effort ladders are now wire-exact and the shifted five-tier effort mapping is gone. Models expose exactly the effort tiers their wire accepts, mapped 1:1: GPT-5.6+ and Anthropic adaptive models with a real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5) expose `low..max`; legacy adaptive models (Opus 4.6 and all Bedrock adaptive) expose `low/medium/high/max`; Sonnet/Haiku 4.6 expose `low/medium/high`; GLM-5.2 on Z.ai/Zhipu/Umans/Ollama Cloud/Baseten and Sakana Fugu and DeepSeek expose `high/max`; local Ollama reasoning models expose `low/medium/high/max`; Fire Pass Kimi exposes `low..max` with distinct `xhigh` and `max` budgets. Removed `SHIFTED_FIVE_TIER_EFFORT_MAP`, `ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER`, and the per-host `xhigh -> "max"` alias maps; selecting a tier a model lacks clamps down via `clampThinkingLevelForModel`. Devin effort routing is now 1:1 onto per-tier siblings (`max -> -max`; families without a `-max` sibling top out at `xhigh`; the fake `minimal -> -low` fallback is gone). Regenerated `models.json`. +- Changed `fillThinkingWireDefaults` to re-derive the effort map whenever the model-defined ladder disagrees with cached metadata, so stale cached surfaces from the shifted-map era normalize to the new wire-exact shape on every `buildModel`. ## [16.3.15] - 2026-07-09 diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index dcfcd3df9..a01430b33 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -71,31 +71,11 @@ const DSML_HEALING_PROVIDERS = new Set([ "openrouter", ]); -/** - * Ollama's OpenAI-compatible `reasoning.effort` only accepts - * `high|medium|low|max|none`; OMP's `minimal`/`xhigh` levels make the server - * reject the turn with HTTP 400 `invalid reasoning value`. Map the two - * unsupported levels onto the closest accepted ones. Stamped in the compat - * builder (not only at discovery) so stale-cached and custom `ollama`-provider - * specs are backfilled on every `buildModel`, not just on a fresh - * `omp models refresh`. Custom OpenAI-compatible providers pointed at a local - * Ollama port under a different provider id are not covered — they must set - * `compat.reasoningEffortMap` themselves. - */ -const OLLAMA_REASONING_EFFORT_MAP: ResolvedOpenAISharedCompat["reasoningEffortMap"] = { minimal: "low", xhigh: "max" }; - -/** - * Merge the Ollama default effort map under any explicit overrides (overrides - * win). No-op off the local `ollama` provider or for non-reasoning models. - */ -function mergeOllamaReasoningEffortMap( - compat: ResolvedOpenAISharedCompat, - provider: string, - reasoning: boolean | undefined, -): void { - if (provider !== "ollama" || !reasoning) return; - compat.reasoningEffortMap = { ...OLLAMA_REASONING_EFFORT_MAP, ...compat.reasoningEffortMap }; -} +// Ollama's OpenAI-compatible `reasoning.effort` accepts `high|medium|low|max|none`; +// `ollama`-provider reasoning models carry the wire-exact `low..max` effort +// ladder (see getModelDefinedEfforts), so no compat-level remapping is needed. +// Custom OpenAI-compatible providers pointed at a local Ollama port under a +// different provider id must set `compat.reasoningEffortMap` themselves. function resolveReasoningDisableMode( thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"], @@ -554,7 +534,6 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; } - mergeOllamaReasoningEffortMap(compat, provider, spec.reasoning); mergeMimoReasoningEffortMap(compat, isMimoReasoningEffortModel); const whenThinkingPolicy = @@ -568,7 +547,6 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv if (whenThinkingPolicy.omitReasoningEffort === undefined && !variant.supportsReasoningEffort) { variant.omitReasoningEffort = true; } - mergeOllamaReasoningEffortMap(variant, provider, spec.reasoning); mergeMimoReasoningEffortMap(variant, isMimoReasoningEffortModel); compat.whenThinking = variant; } @@ -678,7 +656,6 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; } - mergeOllamaReasoningEffortMap(compat, spec.provider, spec.reasoning); return compat; } diff --git a/packages/catalog/src/effort.ts b/packages/catalog/src/effort.ts index 831a13ede..e3491c2a9 100644 --- a/packages/catalog/src/effort.ts +++ b/packages/catalog/src/effort.ts @@ -5,6 +5,7 @@ export const enum Effort { Medium = "medium", High = "high", XHigh = "xhigh", + Max = "max", } export const THINKING_EFFORTS: readonly Effort[] = [ @@ -13,4 +14,5 @@ export const THINKING_EFFORTS: readonly Effort[] = [ Effort.Medium, Effort.High, Effort.XHigh, + Effort.Max, ]; diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 611eb1f60..f2842e4d5 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -61,12 +61,34 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]; const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High]; const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High]; -const GLM_52_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.XHigh]; - -const FUGU_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.XHigh]; -const FUGU_REASONING_EFFORT_MAP: Readonly = { - [Effort.XHigh]: "max", -}; +/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek. */ +const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max]; +/** OpenRouter's DeepSeek route accepts only `high`. */ +const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High]; +/** + * Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models + * with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the + * Fire Pass Kimi router (distinct xhigh and max budgets). + */ +const FIVE_TIER_EFFORTS_LOW_TO_MAX: readonly Effort[] = [ + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + Effort.Max, +]; +/** Legacy adaptive scale (Opus/Sonnet 4.6, every Bedrock adaptive model): four wire tiers, no xhigh. */ +const FOUR_TIER_EFFORTS_LOW_TO_MAX: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.Max]; +/** GLM-5.2 resellers that pass the default lower tiers verbatim and expose the genuine `max` top tier. */ +const DEFAULT_REASONING_EFFORTS_WITH_MAX: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.Max, +]; +/** Local Ollama wire vocabulary (`low`/`medium`/`high`/`max`; `none` is thinking-off). */ +const OLLAMA_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.Max]; type EffortMap = Partial>; const GROQ_QWEN3_32B_REASONING_EFFORT_MAP: Readonly = { @@ -76,57 +98,14 @@ const GROQ_QWEN3_32B_REASONING_EFFORT_MAP: Readonly = { [Effort.High]: "default", [Effort.XHigh]: "default", }; -const DEEPSEEK_REASONING_EFFORT_MAP: Readonly = { - [Effort.Minimal]: "high", - [Effort.Low]: "high", - [Effort.Medium]: "high", - [Effort.High]: "high", - [Effort.XHigh]: "max", -}; const FIREWORKS_REASONING_EFFORT_MAP: Readonly = { [Effort.Minimal]: "none", }; -const ZAI_GLM_52_REASONING_EFFORT_MAP: Readonly = { - [Effort.Minimal]: "none", - [Effort.Low]: "high", - [Effort.Medium]: "high", - [Effort.High]: "high", - [Effort.XHigh]: "max", -}; -const GLM_52_XHIGH_MAX_EFFORT_MAP: Readonly = { - [Effort.XHigh]: "max", -}; const MIMO_REASONING_EFFORT_MAP: Readonly = { [Effort.Minimal]: "low", [Effort.XHigh]: "high", }; -/** - * Effort → wire-value map for a shifted five-tier scale (`low..max`): - * user-facing efforts shift up one notch so the top tier reaches the genuine - * "max" and "high" lands on the recommended "xhigh" coding/agentic default. - * Used by Anthropic adaptive models with a real xhigh tier (Opus 4.7+ and - * Fable/Mythos 5 on the Messages API) and by GPT-5.6+ wire-effort models, - * which expose the same genuine `max` tier above `xhigh`. - */ -export const SHIFTED_FIVE_TIER_EFFORT_MAP: Readonly>> = { - [Effort.Minimal]: "low", - [Effort.Low]: "medium", - [Effort.Medium]: "high", - [Effort.High]: "xhigh", - [Effort.XHigh]: "max", -}; - -/** - * Effort → wire-value map for the legacy 4-tier adaptive scale (Opus 4.6, - * Sonnet 4.6+, and every adaptive model on Bedrock Converse). `low..high` pass - * through verbatim; there is no real "xhigh", so it aliases the top "max" tier. - */ -export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER: Readonly>> = { - [Effort.Minimal]: "low", - [Effort.XHigh]: "max", -}; - const MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP: Readonly = { [Effort.Low]: "adaptive", [Effort.Medium]: "adaptive", @@ -172,8 +151,10 @@ export function resolveModelThinking( /** * Backfill identity-derived wire facts onto explicit thinking metadata. * Explicit `effortMap` / `supportsDisplay` (including `false`) win, except - * model-defined effort restrictions still normalize stale cached capability - * surfaces before request-time code can observe them. + * when the model-defined effort ladder disagrees with the cached surface: + * then both the ladder AND the wire map are re-derived from identity, so + * stale cached metadata from before a wire-truth change (e.g. the retired + * shifted five-tier maps) cannot survive normalization. */ function fillThinkingWireDefaults( spec: ModelSpec, @@ -184,11 +165,9 @@ function fillThinkingWireDefaults( const normalizedEfforts = getModelDefinedEfforts(spec, compat) ?? thinking.efforts; const effortsChanged = !sameEffortList(normalizedEfforts, thinking.efforts); const effortMap = - thinking.effortMap === undefined - ? inferEffortMap(spec, compat, parsed, thinking.mode, normalizedEfforts) - : effortsChanged - ? filterEffortMapToSupportedEfforts(thinking.effortMap, normalizedEfforts) - : undefined; + thinking.effortMap === undefined || effortsChanged + ? inferEffortMap(spec, compat, thinking.mode, normalizedEfforts) + : undefined; const shouldReplaceEffortMap = thinking.effortMap === undefined ? effortMap !== undefined : effortsChanged; const needsDisplay = thinking.supportsDisplay === undefined && @@ -229,7 +208,7 @@ export function deriveThinking(spec: ModelSpec, compat: mode: inferThinkingControlMode(spec, parsed), efforts, }; - const effortMap = inferEffortMap(spec, compat, parsed, config.mode, config.efforts); + const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts); if (effortMap !== undefined) { config.effortMap = effortMap; } @@ -264,11 +243,10 @@ function omitsWireReasoningEffort(api: Api, compat: CompatOf): boolean { function inferEffortMap( spec: ModelSpec, compat: CompatOf, - parsedModel: ParsedModel, mode: ThinkingConfig["mode"], efforts: readonly Effort[], ): EffortMap | undefined { - const detected = inferDetectedEffortMap(spec, compat, parsedModel, mode); + const detected = inferDetectedEffortMap(spec, compat, mode); const configured = readCompatEffortMap(compat); const merged = detected === undefined ? configured : configured === undefined ? detected : { ...detected, ...configured }; @@ -300,9 +278,9 @@ function isOpenAICompatReasoningApi(api: Api): boolean { /** * GPT-5.6+ addressed through a wire `reasoning.effort`/`reasoning_effort` - * field, where the shifted five-tier map applies. Devin (`devin-agent`) - * selects effort by routing to per-tier sibling model ids instead and must - * stay unmapped. + * field, where the five-tier `low..max` wire scale applies. Devin + * (`devin-agent`) selects effort by routing to per-tier sibling model ids + * instead and must stay unmapped. */ function isGpt56PlusWireEffortModel(spec: ModelSpec): boolean { switch (spec.api) { @@ -324,24 +302,64 @@ function getModelDefinedEfforts( compat: CompatOf, ): readonly Effort[] | undefined { if (isGlm52ReasoningEffortModelId(spec.id)) { - // Z.ai/Zhipu and OpenRouter both surface GLM-5.2's full effort ladder, - // including the top `xhigh` (= "max") tier; Umans and Ollama Cloud - // expose only high/max. - if (isZaiThinkingFormat(compat) || isOpenRouterThinkingFormat(compat)) { + // GLM-5.2's reasoning_effort dialect is host-specific (verified against + // live endpoints): + // - Z.ai/Zhipu ("zai" dialect) expose only high/max ("none" is the + // thinking-off state, not a user tier). + // - Umans, Ollama Cloud, and Baseten serve the same two-tier + // high/max scale on their GLM-5.2 routes. + // - OpenRouter rejects `max` — `xhigh` IS its top tier. + // - Other openai-compat hosts (Fireworks, resellers) pass the + // default lower tiers through verbatim and expose the genuine + // `max` above `high` (host quirks like Fireworks' minimal→none + // stay in the host maps). + if (isOpenRouterThinkingFormat(compat)) { return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; } - if (isUmansGlm52ReasoningEffortModel(spec) || isOllamaCloudGlm52ReasoningEffortModel(spec)) { - return GLM_52_HIGH_MAX_REASONING_EFFORTS; + if ( + isZaiThinkingFormat(compat) || + isUmansGlm52ReasoningEffortModel(spec) || + isOllamaCloudGlm52ReasoningEffortModel(spec) || + spec.provider === "baseten" + ) { + return HIGH_MAX_REASONING_EFFORTS; + } + if (isOpenAICompatReasoningApi(spec.api)) { + return DEFAULT_REASONING_EFFORTS_WITH_MAX; } } if (isSakanaFuguReasoningModel(spec)) { - return FUGU_REASONING_EFFORTS; + return HIGH_MAX_REASONING_EFFORTS; } if (isGpt56PlusWireEffortModel(spec)) { - // Normalize stale baked/discovered `low..xhigh` surfaces to the full - // five-tier ladder so the shifted map keeps the native `low` tier - // reachable (user `minimal`). - return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + // Normalize stale baked/discovered `low..xhigh` surfaces to the + // wire-exact five-tier `low..max` ladder. + return FIVE_TIER_EFFORTS_LOW_TO_MAX; + } + const anthropicAdaptive = getAnthropicAdaptiveEfforts(spec); + if (anthropicAdaptive !== undefined) { + return anthropicAdaptive; + } + // Fire Pass's Kimi router accepts low..max with distinct xhigh and max + // budgets; user minimal has no wire tier there. + if (spec.provider === "firepass") { + return FIVE_TIER_EFFORTS_LOW_TO_MAX; + } + // Local Ollama's effort vocabulary is low/medium/high/max regardless of + // model. Custom OpenAI-compatible providers pointed at an Ollama port + // under a different provider id must set `compat.reasoningEffortMap` + // themselves. + if (spec.provider === "ollama") { + return OLLAMA_REASONING_EFFORTS; + } + if (isOpenAICompatReasoningApi(spec.api) && isDeepseekReasoningModel(spec)) { + // DeepSeek's reasoning_effort accepts only high/max; OpenRouter's + // DeepSeek route tops out at high. + return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS; + } + if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) { + // Baseten's gpt-oss router mirrors its GLM route: high/max only. + return HIGH_MAX_REASONING_EFFORTS; } return isOpenAICompatReasoningApi(spec.api) && (isMinimaxM2FamilyModelId(spec.id) || @@ -351,6 +369,27 @@ function getModelDefinedEfforts( : undefined; } +/** + * Wire-exact effort ladders for Anthropic adaptive models (4.6+). Model-defined + * so stale cached surfaces normalize on every build: Messages-API models with + * the real xhigh tier (4.7+) expose the full five-tier `low..max` scale; + * Opus/Sonnet 4.6 and every Bedrock adaptive model stay on the four-tier + * `low/medium/high/max` scale. + */ +function getAnthropicAdaptiveEfforts(spec: ModelSpec): readonly Effort[] | undefined { + const parsed = parseAnthropicModel(bareModelId(spec.id)); + if (!parsed || !isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; + if (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") { + return anthropicModelHasRealXHighEffort(spec, parsed) + ? FIVE_TIER_EFFORTS_LOW_TO_MAX + : FOUR_TIER_EFFORTS_LOW_TO_MAX; + } + if (isOpenRouterAnthropicAdaptiveReasoningModel(parsed, spec)) { + return isAnthropicAdaptiveGenAtLeast(parsed, "4.7") ? FIVE_TIER_EFFORTS_LOW_TO_MAX : FOUR_TIER_EFFORTS_LOW_TO_MAX; + } + return undefined; +} + function isOllamaCloudGlm52ReasoningEffortModel(spec: ModelSpec): boolean { return spec.api === "ollama-chat" && spec.provider === "ollama-cloud" && isGlm52ReasoningEffortModelId(spec.id); } @@ -395,62 +434,31 @@ function isZaiThinkingFormat(compat: CompatOf): boolean { function inferDetectedEffortMap( spec: ModelSpec, compat: CompatOf, - parsedModel: ParsedModel, mode: ThinkingConfig["mode"], ): EffortMap | undefined { if (mode === "anthropic-adaptive") { if (isMinimaxReasoningModelOnAnthropicEndpoint(spec)) { return MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP; } - return anthropicModelHasRealXHighEffort(spec, parsedModel) - ? SHIFTED_FIVE_TIER_EFFORT_MAP - : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; - } - // GLM-5.2 coding SKUs accept `reasoning_effort`, but the effort dialect is - // host-specific (verified against live endpoints): - // - Z.ai/Zhipu ("zai" dialect): the model exposes only none/high/max, so - // `xhigh` 400s — collapse minimal->none, low/medium/high->high, xhigh->max. - // - OpenRouter: `max` 400s and `xhigh` IS its max tier, so it passes `xhigh` - // through literally (no map; the tier is exposed via getModelDefinedEfforts). - // - Umans and Ollama Cloud expose only high/max on their GLM-5.2 routes. - // - Other openai-compat hosts (Fireworks, resellers) keep their distinct - // lower tiers and host quirks (e.g. Fireworks rejects `minimal`, so - // `minimal->none` stays) and only remap the top `xhigh` UI tier onto the - // genuine `max` budget. Filtered to supported efforts later. - const isGlm52 = isGlm52ReasoningEffortModelId(spec.id); - if (isGlm52 && isZaiThinkingFormat(compat)) { - return ZAI_GLM_52_REASONING_EFFORT_MAP; - } - if (isUmansGlm52ReasoningEffortModel(spec) || isOllamaCloudGlm52ReasoningEffortModel(spec)) { - return GLM_52_XHIGH_MAX_EFFORT_MAP; - } - if (isSakanaFuguReasoningModel(spec)) { - return FUGU_REASONING_EFFORT_MAP; - } - if (isGpt56PlusWireEffortModel(spec)) { - return SHIFTED_FIVE_TIER_EFFORT_MAP; + // Adaptive effort ladders are wire-exact (see + // getAnthropicAdaptiveEfforts) — no mapping needed. + return undefined; } if (!isOpenAICompatReasoningApi(spec.api)) { return undefined; } - let map: EffortMap | undefined; if (spec.provider === "groq" && spec.id === "qwen/qwen3-32b") { - map = GROQ_QWEN3_32B_REASONING_EFFORT_MAP; - } else if (isDeepseekReasoningModel(spec)) { - map = DEEPSEEK_REASONING_EFFORT_MAP; - } else if (isOpenAICompatMimoReasoningEffortModel(spec, compat)) { - map = MIMO_REASONING_EFFORT_MAP; - } else if (modelMatchesHost(spec, "openrouter")) { - map = getOpenRouterAnthropicReasoningEffortMap(spec.id); - } else if (modelMatchesHost(spec, "fireworks")) { - map = FIREWORKS_REASONING_EFFORT_MAP; + return GROQ_QWEN3_32B_REASONING_EFFORT_MAP; } - // Overlay GLM-5.2's top-tier `xhigh -> max` on the host base map, except on - // OpenRouter (xhigh IS its max tier; `max` 400s there). - if (isGlm52 && !isOpenRouterThinkingFormat(compat)) { - map = { ...map, ...GLM_52_XHIGH_MAX_EFFORT_MAP }; + if (isOpenAICompatMimoReasoningEffortModel(spec, compat)) { + return MIMO_REASONING_EFFORT_MAP; } - return map; + // Host quirk: Fireworks rejects `minimal` (maps to `none`) on ladders + // that genuinely include it. Filtered to supported efforts later. + if (modelMatchesHost(spec, "fireworks")) { + return FIREWORKS_REASONING_EFFORT_MAP; + } + return undefined; } function isSakanaFuguReasoningModel(spec: ModelSpec): boolean { @@ -471,17 +479,6 @@ function isDeepseekReasoningModel(spec: ModelSpec): bool ); } -function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap | undefined { - const parsed = parseAnthropicModel(bareModelId(modelId)); - if (!parsed) return undefined; - // Adaptive efforts on OpenRouter's completions front: Fable/Mythos, Sonnet 5+, - // and Opus 4.6+ only — older Sonnet versions stay on the plain effort vocabulary there. - if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; - - const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); - return hasRealXHigh ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; -} - function inferSupportedEfforts( parsedModel: ParsedModel, spec: ModelSpec, @@ -507,10 +504,9 @@ function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] { if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) { return GPT_5_1_CODEX_MINI_EFFORTS; } - // 5.6+ exposes the full five-tier ladder: the shifted wire map spans - // low..max, with user `minimal` reaching the native `low` tier. + // 5.6+ exposes the wire-exact five-tier ladder low..max. if (semverGte(model.version, "5.6")) { - return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + return FIVE_TIER_EFFORTS_LOW_TO_MAX; } if (semverGte(model.version, "5.2")) { return GPT_5_2_PLUS_EFFORTS; @@ -554,16 +550,18 @@ function inferAnthropicSupportedEfforts( spec: ModelSpec, compat: CompatOf, ): readonly Effort[] { - if ( - (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") && - semverGte(parsedModel.version, "4.6") - ) { - return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6") - ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH - : DEFAULT_REASONING_EFFORTS; + // Ladders for adaptive-generation models (Opus 4.6+, Sonnet 5+, + // Fable/Mythos) are model-defined and already resolved by + // getAnthropicAdaptiveEfforts. Every other 4.6+ model on the Messages + // API (Sonnet/Haiku 4.6) still runs adaptive mode with the three-tier + // low/medium/high wire scale — no minimal, no max. + if (spec.api === "anthropic-messages" && semverGte(parsedModel.version, "4.6")) { + return LOW_MEDIUM_HIGH_REASONING_EFFORTS; } - if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, spec)) { - return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + // Non-adaptive 4.6 models on Bedrock stay budget-mode, where minimal is + // a legitimate synthetic budget tier. + if (spec.api === "bedrock-converse-stream" && semverGte(parsedModel.version, "4.6")) { + return DEFAULT_REASONING_EFFORTS; } return inferFallbackEfforts(spec, compat); } @@ -752,6 +750,7 @@ export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW return "MEDIUM"; case Effort.High: case Effort.XHigh: + case Effort.Max: return "HIGH"; } } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index b19a10fd5..474e62114 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -1573,8 +1573,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 123000, + "maxTokens": 16000 }, "baidu/ernie-5-0-thinking-latest": { "id": "baidu/ernie-5-0-thinking-latest", @@ -2409,19 +2409,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -2445,19 +2435,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-chat-v3-0324": { @@ -2500,19 +2480,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1": { @@ -2536,19 +2506,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "elevenlabs/eleven_multilingual_v2": { @@ -7738,16 +7698,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic.claude-opus-4-7": { @@ -7772,16 +7727,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -7807,16 +7757,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -7842,16 +7787,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -7906,16 +7846,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "au.anthropic.claude-opus-4-8": { @@ -7940,16 +7875,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8033,16 +7963,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8330,16 +8255,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8481,16 +8401,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "eu.anthropic.claude-opus-4-7": { @@ -8515,16 +8430,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8550,16 +8460,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8672,16 +8577,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8736,16 +8636,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8829,16 +8724,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "global.anthropic.claude-opus-4-7": { @@ -8863,16 +8753,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8898,16 +8783,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9020,16 +8900,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9124,16 +8999,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9159,16 +9029,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9252,16 +9117,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10256,16 +10116,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10407,16 +10262,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "us.anthropic.claude-opus-4-7": { @@ -10441,16 +10291,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10476,16 +10321,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10598,16 +10438,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11141,19 +10976,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11239,19 +11067,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11447,16 +11268,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4-7": { @@ -11481,19 +11297,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11519,19 +11328,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11667,14 +11469,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5": { @@ -11699,19 +11497,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } } @@ -12679,19 +12470,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "moonshotai/Kimi-K2.5": { @@ -12863,9 +12644,8 @@ "thinking": { "mode": "effort", "efforts": [ - "low", - "medium", - "high" + "high", + "max" ] } }, @@ -13293,19 +13073,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -13451,16 +13224,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4-7": { @@ -13485,19 +13253,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -13523,19 +13284,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -13621,14 +13375,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "anthropic/claude-sonnet-5": { @@ -13653,19 +13403,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -14296,19 +14039,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V4-Pro": { @@ -14332,19 +14065,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/gemma-4-31B-it": { @@ -16059,19 +15782,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -16109,19 +15822,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } } }, @@ -16328,19 +16031,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-7-low", - "low": "claude-opus-4-7-medium", - "medium": "claude-opus-4-7-high", - "high": "claude-opus-4-7-xhigh", - "xhigh": "claude-opus-4-7-max" + "low": "claude-opus-4-7-low", + "medium": "claude-opus-4-7-medium", + "high": "claude-opus-4-7-high", + "xhigh": "claude-opus-4-7-xhigh", + "max": "claude-opus-4-7-max" } }, "requestModelId": "claude-opus-4-7-low" @@ -16368,19 +16071,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-7-low-fast", - "low": "claude-opus-4-7-medium-fast", - "medium": "claude-opus-4-7-high-fast", - "high": "claude-opus-4-7-xhigh-fast", - "xhigh": "claude-opus-4-7-max-fast" + "low": "claude-opus-4-7-low-fast", + "medium": "claude-opus-4-7-medium-fast", + "high": "claude-opus-4-7-high-fast", + "xhigh": "claude-opus-4-7-xhigh-fast", + "max": "claude-opus-4-7-max-fast" } }, "requestModelId": "claude-opus-4-7-low-fast" @@ -16408,19 +16111,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-8-low", - "low": "claude-opus-4-8-medium", - "medium": "claude-opus-4-8-high", - "high": "claude-opus-4-8-xhigh", - "xhigh": "claude-opus-4-8-max" + "low": "claude-opus-4-8-low", + "medium": "claude-opus-4-8-medium", + "high": "claude-opus-4-8-high", + "xhigh": "claude-opus-4-8-xhigh", + "max": "claude-opus-4-8-max" } }, "requestModelId": "claude-opus-4-8-low" @@ -16448,19 +16151,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-8-low-fast", - "low": "claude-opus-4-8-medium-fast", - "medium": "claude-opus-4-8-high-fast", - "high": "claude-opus-4-8-xhigh-fast", - "xhigh": "claude-opus-4-8-max-fast" + "low": "claude-opus-4-8-low-fast", + "medium": "claude-opus-4-8-medium-fast", + "high": "claude-opus-4-8-high-fast", + "xhigh": "claude-opus-4-8-xhigh-fast", + "max": "claude-opus-4-8-max-fast" } }, "requestModelId": "claude-opus-4-8-low-fast" @@ -16917,7 +16620,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -16925,7 +16627,6 @@ ], "effortRouting": { "off": "MODEL_GPT_5_2_NONE", - "minimal": "MODEL_GPT_5_2_LOW", "low": "MODEL_GPT_5_2_LOW", "medium": "MODEL_GPT_5_2_MEDIUM", "high": "MODEL_GPT_5_2_HIGH", @@ -16957,7 +16658,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -16965,7 +16665,6 @@ ], "requiresEffort": true, "effortRouting": { - "minimal": "gpt-5-3-codex-low", "low": "gpt-5-3-codex-low", "medium": "gpt-5-3-codex-medium", "high": "gpt-5-3-codex-high", @@ -16997,7 +16696,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17005,7 +16703,6 @@ ], "requiresEffort": true, "effortRouting": { - "minimal": "gpt-5-3-codex-low-priority", "low": "gpt-5-3-codex-low-priority", "medium": "gpt-5-3-codex-medium-priority", "high": "gpt-5-3-codex-high-priority", @@ -17037,7 +16734,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17045,7 +16741,6 @@ ], "effortRouting": { "off": "gpt-5-4-none", - "minimal": "gpt-5-4-low", "low": "gpt-5-4-low", "medium": "gpt-5-4-medium", "high": "gpt-5-4-high", @@ -17077,7 +16772,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17085,7 +16779,6 @@ ], "effortRouting": { "off": "gpt-5-4-none-priority", - "minimal": "gpt-5-4-low-priority", "low": "gpt-5-4-low-priority", "medium": "gpt-5-4-medium-priority", "high": "gpt-5-4-high-priority", @@ -17117,7 +16810,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17125,7 +16817,6 @@ ], "requiresEffort": true, "effortRouting": { - "minimal": "gpt-5-4-mini-low", "low": "gpt-5-4-mini-low", "medium": "gpt-5-4-mini-medium", "high": "gpt-5-4-mini-high", @@ -17157,7 +16848,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17165,7 +16855,6 @@ ], "effortRouting": { "off": "gpt-5-5-none", - "minimal": "gpt-5-5-low", "low": "gpt-5-5-low", "medium": "gpt-5-5-medium", "high": "gpt-5-5-high", @@ -17197,7 +16886,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17205,7 +16893,6 @@ ], "effortRouting": { "off": "gpt-5-5-none-priority", - "minimal": "gpt-5-5-low-priority", "low": "gpt-5-5-low-priority", "medium": "gpt-5-5-medium-priority", "high": "gpt-5-5-high-priority", @@ -17237,19 +16924,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "effortRouting": { "off": "gpt-5-6-luna-none", - "minimal": "gpt-5-6-luna-low", - "low": "gpt-5-6-luna-medium", - "medium": "gpt-5-6-luna-high", - "high": "gpt-5-6-luna-xhigh", - "xhigh": "gpt-5-6-luna-max" + "low": "gpt-5-6-luna-low", + "medium": "gpt-5-6-luna-medium", + "high": "gpt-5-6-luna-high", + "xhigh": "gpt-5-6-luna-xhigh", + "max": "gpt-5-6-luna-max" } }, "requestModelId": "gpt-5-6-luna-none" @@ -17277,7 +16964,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17285,7 +16971,6 @@ ], "effortRouting": { "off": "gpt-5-6-luna-none-priority", - "minimal": "gpt-5-6-luna-low-priority", "low": "gpt-5-6-luna-low-priority", "medium": "gpt-5-6-luna-medium-priority", "high": "gpt-5-6-luna-high-priority", @@ -17317,19 +17002,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "effortRouting": { "off": "gpt-5-6-sol-none", - "minimal": "gpt-5-6-sol-low", - "low": "gpt-5-6-sol-medium", - "medium": "gpt-5-6-sol-high", - "high": "gpt-5-6-sol-xhigh", - "xhigh": "gpt-5-6-sol-max" + "low": "gpt-5-6-sol-low", + "medium": "gpt-5-6-sol-medium", + "high": "gpt-5-6-sol-high", + "xhigh": "gpt-5-6-sol-xhigh", + "max": "gpt-5-6-sol-max" } }, "requestModelId": "gpt-5-6-sol-none" @@ -17357,7 +17042,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17365,7 +17049,6 @@ ], "effortRouting": { "off": "gpt-5-6-sol-none-priority", - "minimal": "gpt-5-6-sol-low-priority", "low": "gpt-5-6-sol-low-priority", "medium": "gpt-5-6-sol-medium-priority", "high": "gpt-5-6-sol-high-priority", @@ -17397,19 +17080,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "effortRouting": { "off": "gpt-5-6-terra-none", - "minimal": "gpt-5-6-terra-low", - "low": "gpt-5-6-terra-medium", - "medium": "gpt-5-6-terra-high", - "high": "gpt-5-6-terra-xhigh", - "xhigh": "gpt-5-6-terra-max" + "low": "gpt-5-6-terra-low", + "medium": "gpt-5-6-terra-medium", + "high": "gpt-5-6-terra-high", + "xhigh": "gpt-5-6-terra-xhigh", + "max": "gpt-5-6-terra-max" } }, "requestModelId": "gpt-5-6-terra-none" @@ -17437,7 +17120,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17445,7 +17127,6 @@ ], "effortRouting": { "off": "gpt-5-6-terra-none-priority", - "minimal": "gpt-5-6-terra-low-priority", "low": "gpt-5-6-terra-low-priority", "medium": "gpt-5-6-terra-medium-priority", "high": "gpt-5-6-terra-high-priority", @@ -17812,15 +17493,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "none" - } + "xhigh", + "max" + ] } } }, @@ -17846,19 +17524,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] }, "compat": { "supportsToolChoice": false, @@ -17886,19 +17554,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] }, "compat": { "supportsToolChoice": false, @@ -18059,11 +17717,10 @@ "low", "medium", "high", - "xhigh" + "max" ], "effortMap": { - "minimal": "none", - "xhigh": "max" + "minimal": "none" } } }, @@ -18563,19 +18220,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -18675,16 +18325,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4.7": { @@ -18713,19 +18358,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -18755,19 +18393,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -18865,14 +18496,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5": { @@ -18901,19 +18528,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -21962,16 +21582,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4-7@default": { @@ -21996,19 +21611,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -22034,19 +21642,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -22102,14 +21703,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5@default": { @@ -22134,19 +21731,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -22171,19 +21761,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.2-maas": { @@ -22207,19 +21787,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "gemini-2.5-flash": { @@ -22746,19 +22316,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "gemma2-9b-it": { @@ -23177,19 +22737,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-R1-0528": { @@ -23213,19 +22763,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3.1": { @@ -23268,19 +22808,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V4-Flash": { @@ -23304,19 +22834,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V4-Pro": { @@ -23340,19 +22860,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/gemma-4-26B-A4B-it": { @@ -26130,8 +25640,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 123000, + "maxTokens": 16000 }, "baidu/qianfan-ocr-fast": { "id": "baidu/qianfan-ocr-fast", @@ -26540,19 +26050,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1": { @@ -26576,19 +26076,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528": { @@ -26612,19 +26102,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-distill-llama-70b": { @@ -26686,19 +26166,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.1-terminus:exacto": { @@ -26741,19 +26211,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -26777,19 +26237,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-speciale": { @@ -26832,19 +26282,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash:discounted": { @@ -26906,19 +26346,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro:discounted": { @@ -28554,8 +27984,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 32000 }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", @@ -28746,8 +28176,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 65535, + "maxTokens": 8000 }, "minimax/minimax-01": { "id": "minimax/minimax-01", @@ -34446,7 +33876,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null + "maxTokens": 262144 }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", @@ -34513,7 +33943,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null + "maxTokens": 262144 }, "thedrummer/cydonia-24b-v4.1": { "id": "thedrummer/cydonia-24b-v4.1", @@ -35579,11 +35009,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm-5v-turbo": { @@ -40546,19 +39973,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3.1": { @@ -40639,19 +40056,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - }, "effortRouting": { "off": "deepseek-ai/deepseek-v3.2-exp", "minimal": "deepseek-ai/deepseek-v3.2-exp-thinking", @@ -40840,19 +40247,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-prover-v2-671b": { @@ -40933,19 +40330,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash": { @@ -40969,19 +40356,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash:thinking": { @@ -41005,19 +40382,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro": { @@ -41041,19 +40408,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro-cheaper": { @@ -41077,19 +40434,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro-cheaper:thinking": { @@ -41113,19 +40460,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro:thinking": { @@ -41149,19 +40486,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "dmind/dmind-1": { @@ -45719,8 +45046,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 32000 }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", @@ -45835,8 +45162,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 65535, + "maxTokens": 8000 }, "MiniMax-M1": { "id": "MiniMax-M1", @@ -46134,8 +45461,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 1000000, + "maxTokens": 40000 }, "miromind-ai/mirothinker-v1.5-235b": { "id": "miromind-ai/mirothinker-v1.5-235b", @@ -48640,19 +47967,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-luna-pro": { @@ -48696,19 +48016,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol-pro": { @@ -48752,19 +48065,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra-pro": { @@ -51265,8 +50571,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 8192, + "maxTokens": 32000 }, "Sao10K/L3.1-70B-Euryale-v2.2": { "id": "Sao10K/L3.1-70B-Euryale-v2.2", @@ -51994,19 +51300,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "TEE/deepseek-v4-pro:thinking": { @@ -52030,19 +51326,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "TEE/gemma-3-27b-it": { @@ -52317,11 +51603,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "TEE/gpt-oss-120b": { @@ -52768,7 +52051,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null + "maxTokens": 262144 }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", @@ -53024,8 +52307,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32000, + "maxTokens": 32000 }, "THUDM/GLM-4-9B-0414": { "id": "THUDM/GLM-4-9B-0414", @@ -54774,11 +54057,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "zai-org/glm-latest": { @@ -55017,19 +54297,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528": { @@ -55054,19 +54324,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528-qwen3-8b": { @@ -55111,19 +54371,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-turbo": { @@ -55148,19 +54398,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1/community": { @@ -55185,19 +54425,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3-0324": { @@ -55262,19 +54492,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.1-terminus": { @@ -55299,19 +54519,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2": { @@ -55336,19 +54546,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -55373,19 +54573,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3/community": { @@ -55430,19 +54620,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro": { @@ -55467,19 +54647,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "dev/glm46": { @@ -57615,11 +56785,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "zai-org/glm-5v-turbo": { @@ -57885,19 +57052,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.1": { @@ -57921,19 +57078,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.1-terminus": { @@ -57957,19 +57104,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.2": { @@ -57993,19 +57130,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v4-flash": { @@ -58029,19 +57156,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v4-pro": { @@ -58065,19 +57182,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/codegemma-1.1-7b": { @@ -58568,8 +57675,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 32000 }, "meta/llama-3.2-90b-vision-instruct": { "id": "meta/llama-3.2-90b-vision-instruct", @@ -61082,11 +60189,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm4.7": { @@ -61619,11 +60723,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "gpt-oss:120b": { @@ -63271,19 +62372,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-luna": { @@ -63309,19 +62403,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-luna-pro": { @@ -63343,26 +62430,19 @@ }, "contextWindow": 1050000, "maxTokens": 128000, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", @@ -63387,19 +62467,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-sol-pro": { @@ -63421,26 +62494,19 @@ }, "contextWindow": 1050000, "maxTokens": 128000, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", @@ -63465,19 +62531,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-terra-pro": { @@ -63499,26 +62558,19 @@ }, "contextWindow": 1050000, "maxTokens": 128000, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "o1": { "id": "o1", @@ -64330,19 +63382,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-luna-pro": { @@ -64372,26 +63417,19 @@ "preferWebsockets": true, "useResponsesLite": true, "priority": 3, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", @@ -64424,19 +63462,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-sol-pro": { @@ -64466,26 +63497,19 @@ "preferWebsockets": true, "useResponsesLite": true, "priority": 1, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", @@ -64518,19 +63542,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-terra-pro": { @@ -64560,26 +63577,19 @@ "preferWebsockets": true, "useResponsesLite": true, "priority": 2, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } } }, "opencode": { @@ -64757,19 +63767,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -64799,19 +63799,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "glm-5": { @@ -64897,11 +63887,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "kimi-k2.5": { @@ -65350,19 +64337,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "claude-3-5-haiku": { @@ -65407,19 +64384,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65535,16 +64505,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4-7": { @@ -65569,19 +64534,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65607,19 +64565,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65705,14 +64656,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5": { @@ -65737,19 +64684,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65774,19 +64714,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-flash-free": { @@ -65810,19 +64740,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -65846,19 +64766,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "gemini-3-flash": { @@ -66118,11 +65028,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "gpt-5": { @@ -68094,19 +67001,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-haiku-4.5": { @@ -68247,16 +67147,11 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.6-fast": { @@ -68281,16 +67176,11 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.7": { @@ -68315,19 +67205,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-opus-4.7-fast": { @@ -68352,19 +67235,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-opus-4.8": { @@ -68389,19 +67265,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-opus-4.8-fast": { @@ -68426,19 +67295,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-sonnet-4": { @@ -68550,19 +67412,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "arcee-ai/trinity-large-preview": { @@ -69091,17 +67946,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-r1": { @@ -69125,17 +67971,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-r1-0528": { @@ -69159,17 +67996,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.1-terminus": { @@ -69193,17 +68021,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.1-terminus:exacto": { @@ -69227,17 +68046,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.2": { @@ -69261,17 +68071,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -69295,17 +68096,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v4-flash": { @@ -69329,17 +68121,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v4-flash:free": { @@ -69363,17 +68146,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v4-pro": { @@ -69397,17 +68171,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "essentialai/rnj-1-instruct": { @@ -73011,19 +71776,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-luna-pro": { @@ -73048,19 +71806,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol": { @@ -73085,19 +71836,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol-pro": { @@ -73122,19 +71866,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra": { @@ -73159,19 +71896,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra-pro": { @@ -73196,19 +71926,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-audio": { @@ -75714,7 +74437,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -75868,17 +74591,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "tngtech/tng-r1t-chimera": { @@ -76815,13 +75529,13 @@ "text" ], "cost": { - "input": 0.84, - "output": 2.64, - "cacheRead": 0.156, + "input": 0.77, + "output": 2.42, + "cacheRead": 0.143, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 128000, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -76876,19 +75590,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } } }, @@ -76956,11 +75660,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "compat": { "includeEncryptedReasoning": false, @@ -76989,11 +75690,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "compat": { "includeEncryptedReasoning": false, @@ -77022,11 +75720,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "compat": { "includeEncryptedReasoning": false, @@ -77337,19 +76032,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3": { @@ -77392,19 +76077,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3.1": { @@ -77447,19 +76122,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "essentialai/Rnj-1-Instruct": { @@ -78224,11 +76889,8 @@ "mode": "anthropic-budget-effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "input": [ "text" @@ -78863,19 +77525,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-flash": { @@ -78899,19 +77551,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -78935,19 +77577,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "e2ee-deepseek-v4-flash": { @@ -82488,19 +81120,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -82646,16 +81271,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.7": { @@ -82680,19 +81300,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -82718,19 +81331,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -82816,14 +81422,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "anthropic/claude-sonnet-5": { @@ -82848,19 +81450,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -85843,11 +84438,11 @@ "thinking": { "mode": "budget", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } }, @@ -85873,11 +84468,11 @@ "thinking": { "mode": "budget", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } }, @@ -85903,11 +84498,11 @@ "thinking": { "mode": "budget", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } }, @@ -87470,19 +86065,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -87506,19 +86091,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "GLM-5.1": { @@ -88636,14 +87211,6 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ @@ -88656,6 +87223,14 @@ "effortMap": { "minimal": "low" } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false } }, "grok-4.3": { @@ -88677,14 +87252,6 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ @@ -88697,6 +87264,14 @@ "effortMap": { "minimal": "low" } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false } }, "grok-4.5": { @@ -88737,20 +87312,7 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": false - }, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "low" - } + "omitReasoningEffort": true } }, "grok-build": { @@ -89905,19 +88467,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -89943,19 +88498,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90091,16 +88639,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.7": { @@ -90125,19 +88668,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90163,19 +88699,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90261,14 +88790,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "anthropic/claude-sonnet-5": { @@ -90293,19 +88818,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90331,19 +88849,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90776,19 +89287,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528": { @@ -90812,19 +89313,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-reasoner": { @@ -90848,19 +89339,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - }, "requiresEffort": true } }, @@ -90885,19 +89366,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -90921,19 +89392,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash": { @@ -90957,19 +89418,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash-free": { @@ -90993,19 +89444,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro": { @@ -91029,19 +89470,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro-free": { @@ -91065,19 +89496,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/gemini-2.0-flash": { @@ -93131,19 +91552,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol": { @@ -93168,19 +91582,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra": { @@ -93205,19 +91612,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-image-1.5": { @@ -94060,7 +92460,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -95169,11 +93569,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm-5.2-free": { @@ -95201,11 +93598,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm-5v-turbo": { @@ -95516,19 +93910,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "none", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "glm-5v-turbo": { @@ -95566,4 +93950,4 @@ } } } -} +} \ No newline at end of file diff --git a/packages/catalog/src/provider-models/ollama.ts b/packages/catalog/src/provider-models/ollama.ts index 3708a8f0f..b9248903e 100644 --- a/packages/catalog/src/provider-models/ollama.ts +++ b/packages/catalog/src/provider-models/ollama.ts @@ -25,8 +25,7 @@ type OllamaShowResponse = { const OLLAMA_RETRY_DELAYS_MS = [2_000, 5_000, 10_000]; const OLLAMA_CLOUD_GLM_52_THINKING: ThinkingConfig = { mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.High, Effort.Max], }; function trimTrailingSlash(value: string): string { diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 0b785b6ca..421a6a404 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -622,9 +622,8 @@ const UMANS_REASONING_EFFORT_BY_LEVEL: Record = { medium: Effort.Medium, high: Effort.High, xhigh: Effort.XHigh, - max: Effort.XHigh, + max: Effort.Max, }; -const UMANS_MAX_REASONING_EFFORT_MAP = { [Effort.XHigh]: "max" } as const; const UMANS_DEFAULT_REASONING_EFFORTS = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] as const; const UMANS_VIA_HANDOFF_MODEL_IDS = ["umans-glm-5.1", "umans-glm-5.2"] as const; @@ -689,9 +688,6 @@ function mapUmansThinkingConfig(value: unknown): ThinkingConfig | undefined { mode: umansHasMaxReasoningLevel(value) ? "anthropic-budget-effort" : "budget", efforts, }; - if (thinking.mode === "anthropic-budget-effort") { - thinking.effortMap = UMANS_MAX_REASONING_EFFORT_MAP; - } if (isRecord(value)) { if (value.can_disable === false) { thinking.requiresEffort = true; @@ -2802,18 +2798,13 @@ export function basetenModelManagerOptions( const baseModel = mapWithBundledReference(entry, defaults, reference); + // Baseten's reasoning router accepts only the high/max + // effort tiers for its GLM-5.2 and gpt-oss routes. const isEffortReasoning = defaults.id === "openai/gpt-oss-120b" || defaults.id === "zai-org/GLM-5.2"; const thinking = isEffortReasoning ? { mode: "effort" as const, - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }, + efforts: [Effort.High, Effort.Max], } : undefined; @@ -2936,8 +2927,7 @@ const SAKANA_FUGU_ULTRA_COST = { input: 5, output: 30, cacheRead: 0.5, cacheWrit const SAKANA_FUGU_ULTRA_CONTEXT_WINDOW = 1_000_000; const SAKANA_FUGU_THINKING: ThinkingConfig = { mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.High, Effort.Max], }; const SAKANA_RESPONSES_COMPAT: ModelSpec<"openai-responses">["compat"] = { includeEncryptedReasoning: false, diff --git a/packages/catalog/src/variant-collapse.ts b/packages/catalog/src/variant-collapse.ts index 46d1f9d52..770d7610d 100644 --- a/packages/catalog/src/variant-collapse.ts +++ b/packages/catalog/src/variant-collapse.ts @@ -111,15 +111,12 @@ function thinkingPair(baseId: string, name: string): EffortVariantFamily { }; } -type DevinTierRoutes = Partial>; +type DevinTierRoutes = Partial>; -const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [ - Effort.Minimal, - Effort.Low, - Effort.Medium, - Effort.High, - Effort.XHigh, -]; +/** Devin families with a `-max` sibling: five wire tiers, `low` floor. */ +const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]; +/** Devin families topping out at `-xhigh` (pre-5.6 GPT, 5.6 fast lanes). */ +const DEVIN_FOUR_TIER_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]; function devinTierFamily( id: string, @@ -132,11 +129,7 @@ function devinTierFamily( for (const effort of efforts) { switch (effort) { case Effort.Minimal: - if (routes.minimal) { - routing[effort] = routes.minimal; - } else if (routes.low) { - routing[effort] = routes.low; - } + if (routes.minimal) routing[effort] = routes.minimal; break; case Effort.Low: if (routes.low) routing[effort] = routes.low; @@ -150,11 +143,20 @@ function devinTierFamily( case Effort.XHigh: if (routes.xhigh) routing[effort] = routes.xhigh; break; + case Effort.Max: + if (routes.max) routing[effort] = routes.max; + break; } } - const members = [routes.off, routes.minimal, routes.low, routes.medium, routes.high, routes.xhigh].filter( - (member, index, items): member is string => typeof member === "string" && items.indexOf(member) === index, - ); + const members = [ + routes.off, + routes.minimal, + routes.low, + routes.medium, + routes.high, + routes.xhigh, + routes.max, + ].filter((member, index, items): member is string => typeof member === "string" && items.indexOf(member) === index); return { id, name, @@ -169,11 +171,9 @@ function devinTierFamily( } /** - * GPT-5.6 (Luna/Sol/Terra) adds a genuine `max` tier above `xhigh`, so the - * standard family shifts every user effort up one notch (`minimal` → `-low` - * … `xhigh` → `-max`), mirroring the Opus 4.7+ five-tier mapping. Devin - * serves no `-max-priority` sibling, so the fast family keeps the direct - * `low..xhigh` `-priority` scale. + * GPT-5.6 (Luna/Sol/Terra) serves per-tier siblings for the full five-tier + * `low..max` wire scale; user efforts route 1:1 onto them. Devin serves no + * `-max-priority` sibling, so the fast family tops out at `xhigh`. */ function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): readonly EffortVariantFamily[] { const base = `gpt-5-6-${variant}`; @@ -183,11 +183,11 @@ function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): re name, { off: `${base}-none`, - minimal: `${base}-low`, - low: `${base}-medium`, - medium: `${base}-high`, - high: `${base}-xhigh`, - xhigh: `${base}-max`, + low: `${base}-low`, + medium: `${base}-medium`, + high: `${base}-high`, + xhigh: `${base}-xhigh`, + max: `${base}-max`, }, DEVIN_FIVE_TIER_EFFORTS, ), @@ -201,7 +201,7 @@ function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): re high: `${base}-high-priority`, xhigh: `${base}-xhigh-priority`, }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), ]; } @@ -357,98 +357,54 @@ export const GEMINI_CLI_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }; export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { families: [ - { - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - members: [ - "claude-opus-4-7-low", - "claude-opus-4-7-medium", - "claude-opus-4-7-high", - "claude-opus-4-7-xhigh", - "claude-opus-4-7-max", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-7-low", - [Effort.Low]: "claude-opus-4-7-medium", - [Effort.Medium]: "claude-opus-4-7-high", - [Effort.High]: "claude-opus-4-7-xhigh", - [Effort.XHigh]: "claude-opus-4-7-max", + devinTierFamily( + "claude-opus-4-7", + "Claude Opus 4.7", + { + low: "claude-opus-4-7-low", + medium: "claude-opus-4-7-medium", + high: "claude-opus-4-7-high", + xhigh: "claude-opus-4-7-xhigh", + max: "claude-opus-4-7-max", }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + "claude-opus-4-7-fast", + "Claude Opus 4.7 Fast", + { + low: "claude-opus-4-7-low-fast", + medium: "claude-opus-4-7-medium-fast", + high: "claude-opus-4-7-high-fast", + xhigh: "claude-opus-4-7-xhigh-fast", + max: "claude-opus-4-7-max-fast", }, - }, - { - id: "claude-opus-4-7-fast", - name: "Claude Opus 4.7 Fast", - members: [ - "claude-opus-4-7-low-fast", - "claude-opus-4-7-medium-fast", - "claude-opus-4-7-high-fast", - "claude-opus-4-7-xhigh-fast", - "claude-opus-4-7-max-fast", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-7-low-fast", - [Effort.Low]: "claude-opus-4-7-medium-fast", - [Effort.Medium]: "claude-opus-4-7-high-fast", - [Effort.High]: "claude-opus-4-7-xhigh-fast", - [Effort.XHigh]: "claude-opus-4-7-max-fast", + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + "claude-opus-4-8", + "Claude Opus 4.8", + { + low: "claude-opus-4-8-low", + medium: "claude-opus-4-8-medium", + high: "claude-opus-4-8-high", + xhigh: "claude-opus-4-8-xhigh", + max: "claude-opus-4-8-max", }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + "claude-opus-4-8-fast", + "Claude Opus 4.8 Fast", + { + low: "claude-opus-4-8-low-fast", + medium: "claude-opus-4-8-medium-fast", + high: "claude-opus-4-8-high-fast", + xhigh: "claude-opus-4-8-xhigh-fast", + max: "claude-opus-4-8-max-fast", }, - }, - { - id: "claude-opus-4-8", - name: "Claude Opus 4.8", - members: [ - "claude-opus-4-8-low", - "claude-opus-4-8-medium", - "claude-opus-4-8-high", - "claude-opus-4-8-xhigh", - "claude-opus-4-8-max", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-8-low", - [Effort.Low]: "claude-opus-4-8-medium", - [Effort.Medium]: "claude-opus-4-8-high", - [Effort.High]: "claude-opus-4-8-xhigh", - [Effort.XHigh]: "claude-opus-4-8-max", - }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, - }, - }, - { - id: "claude-opus-4-8-fast", - name: "Claude Opus 4.8 Fast", - members: [ - "claude-opus-4-8-low-fast", - "claude-opus-4-8-medium-fast", - "claude-opus-4-8-high-fast", - "claude-opus-4-8-xhigh-fast", - "claude-opus-4-8-max-fast", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-8-low-fast", - [Effort.Low]: "claude-opus-4-8-medium-fast", - [Effort.Medium]: "claude-opus-4-8-high-fast", - [Effort.High]: "claude-opus-4-8-xhigh-fast", - [Effort.XHigh]: "claude-opus-4-8-max-fast", - }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, - }, - }, + DEVIN_FIVE_TIER_EFFORTS, + ), devinTierFamily( "gpt-5-2", "GPT-5.2", @@ -459,7 +415,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "MODEL_GPT_5_2_HIGH", xhigh: "MODEL_GPT_5_2_XHIGH", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex", @@ -470,7 +426,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high", xhigh: "gpt-5-3-codex-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex-fast", @@ -481,7 +437,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high-priority", xhigh: "gpt-5-3-codex-xhigh-priority", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4", @@ -493,7 +449,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high", xhigh: "gpt-5-4-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-fast", @@ -505,7 +461,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high-priority", xhigh: "gpt-5-4-xhigh-priority", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-mini", @@ -516,7 +472,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-mini-high", xhigh: "gpt-5-4-mini-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5", @@ -528,7 +484,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high", xhigh: "gpt-5-5-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5-fast", @@ -540,7 +496,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high-priority", xhigh: "gpt-5-5-xhigh-priority", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), ...devinGpt56Families("luna", "GPT-5.6 Luna"), ...devinGpt56Families("sol", "GPT-5.6 Sol"), diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index bd32f2f43..2c6da8a51 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -76,8 +76,7 @@ describe("generated model policies", () => { expect(models[0]?.cost.cacheWrite).toBe(6.25); expect(models[1]?.thinking).toEqual({ mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { minimal: "low", xhigh: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], }); expect(models[1]?.cost.cacheRead).toBe(0.5); expect(models[1]?.cost.cacheWrite).toBe(6.25); @@ -103,8 +102,7 @@ describe("generated model policies", () => { expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }); expect(models[0]?.thinking).toEqual({ mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { minimal: "low", low: "medium", medium: "high", high: "xhigh", xhigh: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], supportsDisplay: true, }); }); diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index c353ac16f..48ded780f 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -277,10 +277,9 @@ describe("LiteLLM provider discovery", () => { reasoning: true, thinking: { mode: "effort", - efforts: ["minimal", "low", "medium", "high", "xhigh"], + efforts: ["minimal", "low", "medium", "high", "max"], effortMap: { minimal: "none", - xhigh: "max", }, }, }); @@ -709,10 +708,9 @@ describe("LiteLLM provider discovery", () => { reasoning: true, thinking: { mode: "effort", - efforts: ["minimal", "low", "medium", "high", "xhigh"], + efforts: ["minimal", "low", "medium", "high", "max"], effortMap: { minimal: "none", - xhigh: "max", }, }, }); diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index a21a323fc..ed08575f1 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -196,7 +196,7 @@ describe("model thinking derivation", () => { api: "openai-completions", provider: "deepseek", baseUrl: "https://api.deepseek.com/v1", - compat: { reasoningEffortMap: { xhigh: "max-plus" } }, + compat: { reasoningEffortMap: { max: "max-plus" } }, }); const openRouterAnthropic = createModel({ id: "anthropic/claude-opus-4.7", @@ -212,20 +212,20 @@ describe("model thinking derivation", () => { medium: "default", high: "default", }); - expect(deepseek.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max-plus", - }); - expect(openRouterAnthropic.thinking?.effortMap).toEqual({ - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }); + // DeepSeek's ladder is the wire-exact high/max pair; explicit compat + // overrides still win over the identity wire values. + expect(getSupportedEfforts(deepseek)).toEqual([Effort.High, Effort.Max]); + expect(deepseek.thinking?.effortMap).toEqual({ max: "max-plus" }); + // OpenRouter-hosted Anthropic adaptive models carry the wire-exact + // five-tier ladder with no remapping. + expect(getSupportedEfforts(openRouterAnthropic)).toEqual([ + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + Effort.Max, + ]); + expect(openRouterAnthropic.thinking?.effortMap).toBeUndefined(); }); it("maps GLM-5.2 reasoning effort per host dialect", () => { @@ -248,19 +248,20 @@ describe("model thinking derivation", () => { baseUrl: "https://openrouter.ai/api/v1", }); - // Z.ai dialect: the model only does none/high/max, so the lower tiers - // collapse and the top `xhigh` tier reaches `max`. - expect(zai.thinking?.effortMap).toEqual({ - minimal: "none", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); - // Fireworks keeps its distinct lower tiers and the `minimal -> none` quirk; - // only the top `xhigh` UI tier remaps onto the genuine `max` budget. - expect(getSupportedEfforts(fireworks)).toContain(Effort.XHigh); - expect(fireworks.thinking?.effortMap).toEqual({ minimal: "none", xhigh: "max" }); + // Z.ai dialect: the model only does none/high/max on the wire, so the + // ladder is the honest high/max pair (none = thinking off). + expect(getSupportedEfforts(zai)).toEqual([Effort.High, Effort.Max]); + expect(zai.thinking?.effortMap).toBeUndefined(); + // Fireworks keeps its distinct lower tiers and the `minimal -> none` + // quirk; the genuine `max` tier sits above `high`. + expect(getSupportedEfforts(fireworks)).toEqual([ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.Max, + ]); + expect(fireworks.thinking?.effortMap).toEqual({ minimal: "none" }); // OpenRouter rejects `max` and treats `xhigh` as its max tier: expose the // `xhigh` tier and pass it through unmapped. expect(getSupportedEfforts(openRouter)).toContain(Effort.XHigh); @@ -428,30 +429,34 @@ describe("model thinking derivation", () => { }, }); expect(mapEffortToAnthropicAdaptiveEffort(minimaxM3, Effort.High)).toBe("adaptive"); - // Opus 4.6 has no real xhigh level — the baked 4-tier map aliases XHigh to "max". - expect(opus46.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" }); - expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toBe("max"); - // Opus 4.7+ on the Messages API exposes the full five-tier scale: the baked - // map shifts each user-facing effort up one notch so the top tier reaches "max". - expect(opus47.thinking?.effortMap).toEqual({ - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }); - expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Minimal)).toBe("low"); - expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("xhigh"); - expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max"); - expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh"); - expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max"); - expect(mapEffortToAnthropicAdaptiveEffort(sonnet5, Effort.High)).toBe("xhigh"); - expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.XHigh)).toBe("max"); - // Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max". - expect(opus47Bedrock.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" }); + // Opus 4.6 has no real xhigh tier — the honest ladder is the four-tier + // low/medium/high/max wire scale, mapped 1:1. + expect(getSupportedEfforts(opus46)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(opus46.thinking?.effortMap).toBeUndefined(); + expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.Max)).toBe("max"); + expect(() => mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toThrow(/not supported/); + // Opus 4.7+ on the Messages API exposes the full five-tier wire scale + // low..max with no remapping. + expect(getSupportedEfforts(opus47)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus47.thinking?.effortMap).toBeUndefined(); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Low)).toBe("low"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("high"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("xhigh"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Max)).toBe("max"); + expect(() => mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Minimal)).toThrow(/not supported/); + expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.XHigh)).toBe("xhigh"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5, Effort.Max)).toBe("max"); + // Bedrock Converse stays on the four-tier scale regardless of version. + expect(getSupportedEfforts(opus47Bedrock)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(opus47Bedrock.thinking?.effortMap).toBeUndefined(); expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high"); - expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.High)).toBe("high"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.Max)).toBe("max"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.Max)).toBe("max"); + expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.XHigh)).toThrow(/not supported/); + // Sonnet 4.6 runs adaptive mode on the three-tier low/medium/high scale. + expect(getSupportedEfforts(sonnet46)).toEqual([Effort.Low, Effort.Medium, Effort.High]); expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.XHigh)).toThrow(/not supported/); + expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.Max)).toThrow(/not supported/); }); it("bakes adaptive display support for Opus 4.7+, Sonnet 5+, and Fable/Mythos 5", () => { @@ -488,8 +493,9 @@ describe("model thinking derivation", () => { }); it("backfills wire facts onto explicit thinking, explicit values winning", () => { - // Authored capability surface (mode/efforts) keeps identity-derived wire - // facts: configs never need to know Anthropic's tier tables. + // Authored partial ladders on wire-exact models normalize to the + // model-defined ladder, and the wire map is re-derived alongside: + // stale cached surfaces cannot pin retired wire facts. const filled = createModel({ id: "claude-opus-4-8", api: "anthropic-messages", @@ -498,24 +504,24 @@ describe("model thinking derivation", () => { }); expect(filled.thinking).toEqual({ mode: "anthropic-adaptive", - efforts: [Effort.Low, Effort.High], - effortMap: { low: "medium", high: "xhigh" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], supportsDisplay: true, }); - // Explicit wire facts are authoritative — including `false`. + // Explicit wire facts are authoritative — including `false` — when the + // authored ladder matches the wire truth. const pinned = createModel({ id: "claude-opus-4-8", api: "anthropic-messages", provider: "anthropic", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Low, Effort.High], - effortMap: { xhigh: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + effortMap: { max: "ultra" }, supportsDisplay: false, }, }); - expect(pinned.thinking?.effortMap).toEqual({ xhigh: "max" }); + expect(pinned.thinking?.effortMap).toEqual({ max: "ultra" }); expect(pinned.thinking?.supportsDisplay).toBe(false); }); @@ -568,7 +574,7 @@ describe("model thinking derivation", () => { expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined(); }); - it("bakes the GPT-5.6 shifted five-tier effort map on wire-effort APIs", () => { + it("bakes the wire-exact five-tier low..max ladder on GPT-5.6 wire-effort APIs", () => { const codex = createModel({ id: "gpt-5.6-sol", api: "openai-codex-responses", @@ -577,19 +583,12 @@ describe("model thinking derivation", () => { expect(codex.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }); - // Stale baked four-tier metadata (caches/discovery) normalizes back to - // the five-tier ladder with the map attached — the wire-defaults - // backfill path — and namespaced OpenRouter ids parse. + // Stale baked metadata (caches/discovery) — including shifted-era maps — + // normalizes to the wire-exact ladder with the map re-derived away, and + // namespaced OpenRouter ids parse. const staleOpenRouter = createModel({ id: "openai/gpt-5.6-terra", api: "openrouter", @@ -597,24 +596,24 @@ describe("model thinking derivation", () => { baseUrl: "https://openrouter.ai/api/v1", thinking: { mode: "effort", - efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "low", + low: "medium", + medium: "high", + high: "xhigh", + xhigh: "max", + }, }, }); expect(staleOpenRouter.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }); }); - it("keeps pre-5.6 and Devin-routed GPT models off the shifted effort map", () => { + it("keeps pre-5.6 and Devin-routed GPT models on their own effort surfaces", () => { const gpt55 = createModel({ id: "gpt-5.5", api: "openai-responses", @@ -628,7 +627,7 @@ describe("model thinking derivation", () => { expect(gpt55.thinking?.effortMap).toBeUndefined(); // Devin selects effort by routing to per-tier sibling model ids, never - // via a wire reasoning.effort field — the shifted map must not attach. + // via a wire reasoning.effort field — no effort map may attach. const devin = createModel({ id: "gpt-5-6-sol", api: "devin-agent", @@ -636,20 +635,20 @@ describe("model thinking derivation", () => { baseUrl: "https://server.codeium.com", thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], effortRouting: { off: "gpt-5-6-sol-none", - minimal: "gpt-5-6-sol-low", - low: "gpt-5-6-sol-medium", - medium: "gpt-5-6-sol-high", - high: "gpt-5-6-sol-xhigh", - xhigh: "gpt-5-6-sol-max", + low: "gpt-5-6-sol-low", + medium: "gpt-5-6-sol-medium", + high: "gpt-5-6-sol-high", + xhigh: "gpt-5-6-sol-xhigh", + max: "gpt-5-6-sol-max", }, }, }); expect(devin.thinking?.effortMap).toBeUndefined(); - expect(devin.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + expect(devin.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); }); }); @@ -712,7 +711,7 @@ describe("model thinking runtime helpers", () => { ); }); - it("maps GLM-5.2 xhigh to Z.AI provider-native max", () => { + it("exposes the Z.AI GLM-5.2 high/max wire pair directly", () => { const model = createModel({ id: "glm-5.2", api: "openai-completions", @@ -723,19 +722,15 @@ describe("model thinking runtime helpers", () => { expect(model.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "none", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }, + efforts: [Effort.High, Effort.Max], }); - expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh); + expect(requireSupportedEffort(model, Effort.Max)).toBe(Effort.Max); + expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(/Supported efforts: high, max/); + // Selecting a retired tier clamps down instead of erroring in UI flows. + expect(clampThinkingLevelForModel(model, Effort.XHigh)).toBe(Effort.High); }); - it("maps Ollama Cloud GLM-5.2 xhigh to max and hides unsupported lower efforts", () => { + it("exposes Ollama Cloud GLM-5.2 high/max and hides unsupported lower efforts", () => { const model = createModel({ id: "glm-5.2", api: "ollama-chat", @@ -745,14 +740,11 @@ describe("model thinking runtime helpers", () => { expect(model.thinking).toEqual({ mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { - xhigh: "max", - }, + efforts: [Effort.High, Effort.Max], }); expect(requireSupportedEffort(model, Effort.High)).toBe(Effort.High); - expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh); - expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/Supported efforts: high, xhigh/); + expect(requireSupportedEffort(model, Effort.Max)).toBe(Effort.Max); + expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/Supported efforts: high, max/); }); it("derives binary-thinking fallback from resolved compat when catalog compat is partial", () => { @@ -774,7 +766,7 @@ describe("model thinking runtime helpers", () => { ); }); - it("exposes xhigh for OpenRouter-hosted Anthropic adaptive models", () => { + it("exposes wire-exact adaptive ladders for OpenRouter-hosted Anthropic models", () => { const fable = createModel({ id: "anthropic/claude-fable-5", api: "openai-completions", @@ -795,12 +787,13 @@ describe("model thinking runtime helpers", () => { api: "openai-completions", provider: "openrouter", }); - expect(fable.thinking?.efforts.at(-1)).toBe(Effort.XHigh); - expect(opus46.thinking?.efforts.at(-1)).toBe(Effort.XHigh); + expect(fable.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus46.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); expect(sonnet46.thinking?.efforts.at(-1)).toBe(Effort.High); - expect(sonnet5.thinking?.efforts.at(-1)).toBe(Effort.XHigh); - expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh); + expect(sonnet5.thinking?.efforts.at(-1)).toBe(Effort.Max); + expect(requireSupportedEffort(fable, Effort.Max)).toBe(Effort.Max); expect(requireSupportedEffort(sonnet5, Effort.XHigh)).toBe(Effort.XHigh); + expect(() => requireSupportedEffort(opus46, Effort.XHigh)).toThrow(/not supported/); }); it("enables xhigh for openai-responses and openai-codex-responses APIs", () => { diff --git a/packages/catalog/test/ollama-cloud-provider.test.ts b/packages/catalog/test/ollama-cloud-provider.test.ts index 4f640c145..4f85dac46 100644 --- a/packages/catalog/test/ollama-cloud-provider.test.ts +++ b/packages/catalog/test/ollama-cloud-provider.test.ts @@ -141,8 +141,7 @@ describe("ollama-cloud provider support", () => { expect(model?.reasoning).toBe(true); expect(built?.thinking).toEqual({ mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { xhigh: "max" }, + efforts: [Effort.High, Effort.Max], }); }); @@ -286,7 +285,7 @@ describe("ollama-cloud provider support", () => { expect(result.errorMessage).toContain("prompt filled the context window"); }); - test("sends max for GLM-5.2 xhigh reasoning on Ollama Cloud", async () => { + test("sends native max for GLM-5.2 max reasoning on Ollama Cloud", async () => { let requestBody: Record | undefined; const fetchMock: FetchImpl = vi.fn(async (_input, init) => { requestBody = JSON.parse(String(init?.body ?? "{}")) as Record; @@ -302,7 +301,7 @@ describe("ollama-cloud provider support", () => { provider: "ollama-cloud", baseUrl: "https://ollama.com", reasoning: true, - thinking: { mode: "effort", efforts: [Effort.High, Effort.XHigh], effortMap: { [Effort.XHigh]: "max" } }, + thinking: { mode: "effort", efforts: [Effort.High, Effort.Max] }, input: ["text"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, @@ -315,7 +314,7 @@ describe("ollama-cloud provider support", () => { { apiKey: "cloud-test-key", fetch: fetchMock, - reasoning: Effort.XHigh, + reasoning: Effort.Max, }, ).result(); diff --git a/packages/catalog/test/ollama-provider.test.ts b/packages/catalog/test/ollama-provider.test.ts index acbf10fb3..aebd75789 100644 --- a/packages/catalog/test/ollama-provider.test.ts +++ b/packages/catalog/test/ollama-provider.test.ts @@ -3,6 +3,7 @@ import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama"; import type { Context, Tool } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking"; import { ollamaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; @@ -87,10 +88,14 @@ describe("ollama local provider discovery", () => { const builtReasoningModel = reasoningModel ? buildModel(reasoningModel) : undefined; const builtPlainModel = plainModel ? buildModel(plainModel) : undefined; - // Ollama's OpenAI-compatible endpoint rejects "minimal" with HTTP 400; - // reasoning models must bake a thinking effort map to an accepted level (low). + // Ollama's OpenAI-compatible endpoint accepts low/medium/high/max; + // reasoning models carry that wire-exact ladder with no remapping + // (minimal/xhigh never reach the wire because they are not offered). expect(reasoningModel?.reasoning).toBe(true); - expect(builtReasoningModel?.thinking?.effortMap).toMatchObject({ minimal: "low" }); + expect(builtReasoningModel?.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], + }); // Non-reasoning models never send an effort, so they carry no thinking metadata. expect(plainModel?.reasoning).toBe(false); expect(builtPlainModel?.thinking).toBeUndefined(); @@ -150,7 +155,7 @@ describe("ollama tool forcing", () => { }); }); -describe("ollama reasoning effort backfill (buildModel)", () => { +describe("ollama reasoning effort normalization (buildModel)", () => { const staleOllamaSpec = ( api: TApi, compat?: ModelSpec["compat"], @@ -170,38 +175,32 @@ describe("ollama reasoning effort backfill (buildModel)", () => { compat, }) as ModelSpec; - test("stamps the effort map on a stale ollama responses spec lacking compat", () => { - // A cache row or hand-written config written before the remap existed: - // reasoning-capable, `minimal` offered, but no reasoningEffortMap. The - // builder must backfill it so the wire never sends raw `minimal`/`xhigh`. + test("normalizes a stale ollama responses spec to the wire-exact ladder", () => { + // A cache row or hand-written config from the remap era: reasoning-capable + // with `minimal` offered. The builder must normalize the ladder so the + // wire never sends raw `minimal`/`xhigh`. const model = buildModel(staleOllamaSpec("openai-responses")); - expect(model.compat.reasoningEffortMap).toMatchObject({ minimal: "low", xhigh: "max" }); - // xhigh drops out of thinking.effortMap — it is not an offered effort. - expect(model.thinking?.effortMap).toEqual({ minimal: "low" }); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); + // Retired tiers clamp instead of erroring. + expect(clampThinkingLevelForModel(model, Effort.Minimal)).toBe(Effort.Low); + expect(clampThinkingLevelForModel(model, Effort.XHigh)).toBe(Effort.High); }); - test("backfills openai-completions ollama specs too", () => { + test("normalizes openai-completions ollama specs too", () => { const model = buildModel(staleOllamaSpec("openai-completions")); - expect(model.compat.reasoningEffortMap).toMatchObject({ minimal: "low", xhigh: "max" }); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); }); - test("explicit overrides win while missing ollama defaults stay", () => { - const model = buildModel(staleOllamaSpec("openai-responses", { reasoningEffortMap: { minimal: "medium" } })); - expect(model.compat.reasoningEffortMap).toEqual({ minimal: "medium", xhigh: "max" }); + test("explicit compat overrides survive for live tiers", () => { + const model = buildModel(staleOllamaSpec("openai-responses", { reasoningEffortMap: { high: "medium" } })); + expect(model.compat.reasoningEffortMap).toEqual({ high: "medium" }); + expect(model.thinking?.effortMap).toEqual({ high: "medium" }); }); test("leaves non-ollama providers untouched", () => { const model = buildModel({ ...staleOllamaSpec("openai-responses"), provider: "custom" }); expect(model.compat.reasoningEffortMap).toEqual({}); - }); - - test("merges the ollama defaults into the whenThinking variant", () => { - const model = buildModel( - staleOllamaSpec("openai-completions", { whenThinking: { reasoningEffortMap: { minimal: "medium" } } }), - ); - // The thinking-engaged variant must keep the xhigh default; otherwise a - // partial whenThinking override would re-leak raw `minimal`/`xhigh`. - expect(model.compat.whenThinking?.reasoningEffortMap).toEqual({ minimal: "medium", xhigh: "max" }); - expect(model.compat.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "max" }); + expect(model.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]); }); }); diff --git a/packages/catalog/test/sakana-provider.test.ts b/packages/catalog/test/sakana-provider.test.ts index f7b9d3ac2..0456be443 100644 --- a/packages/catalog/test/sakana-provider.test.ts +++ b/packages/catalog/test/sakana-provider.test.ts @@ -60,8 +60,8 @@ describe("Sakana AI provider support", () => { expect(bundled.find(model => model.id === "fugu-ultra-20260615")?.contextWindow).toBe(1_000_000); for (const model of bundled) { expect(model.api).toBe("openai-responses"); - expect(model.thinking?.efforts).toEqual([Effort.High, Effort.XHigh]); - expect(model.thinking?.effortMap?.[Effort.XHigh]).toBe("max"); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); expect((model.compat as ResolvedOpenAIResponsesCompat).includeEncryptedReasoning).toBe(false); expect((model.compat as ResolvedOpenAIResponsesCompat).streamIdleTimeoutMs).toBe(0); } @@ -95,8 +95,8 @@ describe("Sakana AI provider support", () => { expect(models?.map(model => model.id)).toEqual(["fugu", "fugu-next", "fugu-ultra"]); const fuguNext = models?.find(model => model.id === "fugu-next"); expect(fuguNext?.reasoning).toBe(true); - expect(fuguNext?.thinking?.efforts).toEqual([Effort.High, Effort.XHigh]); - expect(fuguNext?.thinking?.effortMap?.[Effort.XHigh]).toBe("max"); + expect(fuguNext?.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(fuguNext?.thinking?.effortMap).toBeUndefined(); expect(fuguNext?.compat?.includeEncryptedReasoning).toBe(false); }); diff --git a/packages/catalog/test/umans-provider.test.ts b/packages/catalog/test/umans-provider.test.ts index a128d59f8..a641e6ec9 100644 --- a/packages/catalog/test/umans-provider.test.ts +++ b/packages/catalog/test/umans-provider.test.ts @@ -118,12 +118,11 @@ describe("umans provider catalog", () => { thinking: { mode: "anthropic-budget-effort", defaultLevel: "high", - efforts: ["high", "xhigh"], - effortMap: { xhigh: "max" }, + efforts: ["high", "max"], }, }); if (!glm52) throw new Error("Umans GLM 5.2 was not discovered"); - expect(glm52.thinking?.effortMap).toEqual({ [Effort.XHigh]: "max" }); + expect(glm52.thinking?.effortMap).toBeUndefined(); expect(glm52.thinking?.defaultLevel).toBe(Effort.High); }); @@ -307,15 +306,15 @@ describe("umans provider catalog", () => { }); }); - it("bundles Umans GLM 5.2 high/max reasoning metadata with the max wire effort", () => { + it("bundles Umans GLM 5.2 with the wire-exact high/max ladder", () => { const providers = modelsJson as Record>; const model = providers.umans?.["umans-glm-5.2"]; expect(model).toBeDefined(); expect(model.thinking).toMatchObject({ mode: "anthropic-budget-effort", - efforts: ["high", "xhigh"], - effortMap: { xhigh: "max" }, + efforts: ["high", "max"], }); + expect(model.thinking?.effortMap).toBeUndefined(); }); }); diff --git a/packages/catalog/test/variant-collapse.test.ts b/packages/catalog/test/variant-collapse.test.ts index ae373b6bf..cbfc93781 100644 --- a/packages/catalog/test/variant-collapse.test.ts +++ b/packages/catalog/test/variant-collapse.test.ts @@ -17,6 +17,7 @@ import { ANTIGRAVITY_VARIANT_COLLAPSE_TABLE, collapseEffortVariants, collapseEffortVariantsAcrossProviders, + DEVIN_VARIANT_COLLAPSE_TABLE, deriveThinkingPairFamilies, GEMINI_CLI_VARIANT_COLLAPSE_TABLE, getVariantAliasSources, @@ -526,6 +527,45 @@ describe("collapseEffortVariantsAcrossProviders", () => { }); }); +describe("Devin tier routing", () => { + const family = (id: string) => { + const found = DEVIN_VARIANT_COLLAPSE_TABLE.families.find(f => f.id === id); + if (!found) throw new Error(`Devin family ${id} missing`); + return found; + }; + + it("routes user efforts 1:1 onto per-tier siblings including max", () => { + const opus = family("claude-opus-4-8"); + expect(opus.routing).toEqual({ + [Effort.Low]: "claude-opus-4-8-low", + [Effort.Medium]: "claude-opus-4-8-medium", + [Effort.High]: "claude-opus-4-8-high", + [Effort.XHigh]: "claude-opus-4-8-xhigh", + [Effort.Max]: "claude-opus-4-8-max", + }); + expect(opus.thinking.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus.thinking.requiresEffort).toBe(true); + + const sol = family("gpt-5-6-sol"); + expect(sol.routing[Effort.Max]).toBe("gpt-5-6-sol-max"); + expect(sol.routing[Effort.Low]).toBe("gpt-5-6-sol-low"); + expect(sol.routing.off).toBe("gpt-5-6-sol-none"); + expect(sol.routing[Effort.Minimal]).toBeUndefined(); + }); + + it("keeps families without a -max sibling on the xhigh ceiling", () => { + const solFast = family("gpt-5-6-sol-fast"); + expect(solFast.thinking.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + expect(solFast.routing[Effort.Max]).toBeUndefined(); + expect(solFast.routing[Effort.XHigh]).toBe("gpt-5-6-sol-xhigh-priority"); + + const gpt55 = family("gpt-5-5"); + expect(gpt55.thinking.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + expect(gpt55.routing[Effort.Minimal]).toBeUndefined(); + expect(gpt55.routing[Effort.Max]).toBeUndefined(); + }); +}); + describe("variant aliases", () => { it("resolves members and recycled ids per provider", () => { expect(resolveVariantAlias("google-antigravity", "gemini-3.5-flash-low")).toBe("gemini-3.5-flash"); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 658e3a904..214127a96 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,10 +6,18 @@ - Renamed the bundled agent `explore` to `scout` (including renaming its configuration keys, prompt files, and task definitions). Any configurations, allowlists, or invocations referencing `explore` must now use `scout`. +### Added + +- Added `max` as a native, first-class thinking tier for supported models +- Added `thinkingBudgets.max` configuration setting +- Updated terminal theme to support optional `thinkingMax` border color and icons + +- Added a real `max` thinking level above `xhigh` with an optional `thinkingMax` theme border color (falls back to `thinkingXhigh`). `max` owns the top status-line icons (`◉` unicode, fire nerd-font, `[max]` ascii); the nerd-font preset uses an empty-to-full battery ramp for `minimal` through `xhigh` and shuffle while automatic effort is unresolved. `max` appears in cycling, selectors, `--thinking`, `:max` model suffixes, role scopes, settings, and completions on models that genuinely support it; on other models it clamps down like any unsupported tier. + ### Changed - Renamed the bundled agent `explore` to `scout` (including all internal references, prompt definitions, and task tool configurations). Any custom configurations or task invocations referencing `explore` must now use `scout`. - +- Changed `max` from a parse-time alias of `xhigh` to a distinct level everywhere (CLI flag, `:max` suffix, `defaultThinkingLevel`, ACP/RPC): the effort a model receives is now exactly the tier its wire supports, with no shifted remapping. Ultrathink now requests `max` (clamped per model); automatic thinking still tops out at `xhigh`. Added `thinkingBudgets.max` (default 32768). - Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) ### Fixed diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 82c4e0545..7fb1491b6 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -690,7 +690,7 @@ function normalizeSuppressedSelector( const trimmed = selector.trim(); if (!trimmed) return trimmed; const parsed = parseModelString(trimmed, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => hasLiveModel?.(provider, id) === true, }); diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index f09310b51..32ed6116f 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -79,23 +79,24 @@ export interface ScopedModel { } interface ThinkingSuffixOptions { - allowMaxAlias?: boolean; + allowMaxSuffix?: boolean; allowAutoAlias?: boolean; } interface ModelStringParseOptions extends ThinkingSuffixOptions { isLiteralModelId?: (provider: string, id: string) => boolean; } -// Alias-suffix recognition for the model-pattern parser: `:max` maps to xhigh -// and `:auto` maps to the auto sentinel. Both are gated behind the alias flags -// (and the literal-id / exact-match guards on the callers) so a real model id -// ending in `:max` / `:auto` isn't silently reinterpreted as a thinking suffix. -const MAX_THINKING_SUFFIX_OPTIONS: ThinkingSuffixOptions = { allowMaxAlias: true, allowAutoAlias: true }; +// Suffix recognition for the model-pattern parser: `:max` is a real thinking +// level and `:auto` maps to the auto sentinel. Both are gated behind flags +// (and the literal-id / exact-match guards on the callers) because real model +// ids end in `:max` (e.g. `glm-4.7:max`) — an ungated split would silently +// reinterpret them as a thinking suffix. +const MAX_THINKING_SUFFIX_OPTIONS: ThinkingSuffixOptions = { allowMaxSuffix: true, allowAutoAlias: true }; function parseThinkingSuffix(value: string, options?: ThinkingSuffixOptions): ConfiguredThinkingLevel | undefined { const level = parseThinkingLevel(value); + if (level === ThinkingLevel.Max) return options?.allowMaxSuffix === true ? level : undefined; if (level !== undefined) return level; - if (options?.allowMaxAlias === true && value === "max") return ThinkingLevel.XHigh; if (options?.allowAutoAlias === true && value === AUTO_THINKING) return AUTO_THINKING; return undefined; } @@ -104,8 +105,9 @@ function parseThinkingSuffix(value: string, options?: ThinkingSuffixOptions): Co * Split a trailing `:` thinking selector off a model pattern. * * `level` is set when the suffix parses as a concrete thinking level (or, when - * the caller opts in via `allowMaxAlias`/`allowAutoAlias`, the `:max` / `:auto` - * aliases); `base` then has the suffix stripped. Otherwise `base` is the input. + * the caller opts in via `allowMaxSuffix`/`allowAutoAlias`, the guarded `:max` + * level / `:auto` sentinel); `base` then has the suffix stripped. Otherwise + * `base` is the input. * `minColonIndex` requires the colon to appear strictly after that index — * role-alias callers pass `PREFIX_MODEL_ROLE.length` so the base is at least * as long as the `pi/` prefix. @@ -183,7 +185,7 @@ export function parseModelString( // Strip strict thinking level suffixes first (e.g. "claude-sonnet-4-6:high" -> id "claude-sonnet-4-6", thinkingLevel "high"). const strict = splitThinkingSuffix(id); if (strict.level) return { provider, id: strict.base, thinkingLevel: strict.level }; - // `max` is a provider-facing alias for xhigh, but real model IDs can end in + // `max` is a real thinking level, but real model IDs can also end in // `:max`. Context-aware callers pass a literal lookup so those models win. const maxAlias = splitThinkingSuffix(id, -1, options); if (maxAlias.level) { @@ -239,7 +241,7 @@ function getOpenRouterRouteSuffix(modelId: string): { baseId: string; suffix: st } const suffix = modelId.slice(colonIdx + 1).trim(); - // `max` is a thinking-level alias (xhigh), never an OpenRouter route suffix, so + // `max` is a thinking-level suffix, never an OpenRouter route suffix, so // `openrouter/:max` falls through to the max-aware selector split instead of // being cloned into a literal `:max` model id with the reasoning level lost. if (!suffix || parseThinkingSuffix(suffix, MAX_THINKING_SUFFIX_OPTIONS)) { @@ -766,7 +768,7 @@ function parseModelPatternWithContext( // No match - try stripping a valid thinking suffix and recursing. // `max` is accepted only after the full pattern failed, so literal model IDs - // ending in `:max` keep winning over the alias. + // ending in `:max` keep winning over the thinking suffix. const { base, level } = splitThinkingSuffix(pattern, -1, MAX_THINKING_SUFFIX_OPTIONS); if (level) { const result = parseModelPatternWithContext(base, availableModels, context, options); diff --git a/packages/coding-agent/src/config/models-config-schema.ts b/packages/coding-agent/src/config/models-config-schema.ts index 7d886bff5..1309ad8de 100644 --- a/packages/coding-agent/src/config/models-config-schema.ts +++ b/packages/coding-agent/src/config/models-config-schema.ts @@ -23,6 +23,7 @@ const ReasoningEffortMapSchema = type({ "medium?": "string", "high?": "string", "xhigh?": "string", + "max?": "string", }); const OpenAICompatFields = { @@ -74,13 +75,13 @@ const ApiSchema = type( '"openai-completions" | "openai-responses" | "openai-codex-responses" | "azure-openai-responses" | "anthropic-messages" | "google-generative-ai" | "google-gemini-cli" | "google-vertex"', ); -const EffortSchema = type('"minimal" | "low" | "medium" | "high" | "xhigh"'); +const EffortSchema = type('"minimal" | "low" | "medium" | "high" | "xhigh" | "max"'); const ThinkingControlModeSchema = type( '"effort" | "budget" | "google-level" | "anthropic-adaptive" | "anthropic-budget-effort"', ); -const EFFORT_ORDER = ["minimal", "low", "medium", "high", "xhigh"] as const; +const EFFORT_ORDER = ["minimal", "low", "medium", "high", "xhigh", "max"] as const; /** * Accepts the canonical `efforts` vocabulary plus the legacy diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 5ee3269d1..81f4ef7dd 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -936,7 +936,7 @@ export const SETTINGS_SCHEMA = { // Reasoning and prompts defaultThinkingLevel: { type: "enum", - values: [...THINKING_EFFORTS, AUTO_THINKING, "max"], + values: [...THINKING_EFFORTS, AUTO_THINKING], default: "high", ui: { tab: "model", @@ -4964,6 +4964,8 @@ export const SETTINGS_SCHEMA = { "thinkingBudgets.high": { type: "number", default: 16384 }, "thinkingBudgets.xhigh": { type: "number", default: 32768 }, + + "thinkingBudgets.max": { type: "number", default: 32768 }, } as const; // ═══════════════════════════════════════════════════════════════════════════ @@ -5180,6 +5182,7 @@ export interface ThinkingBudgetsSettings { medium: number; high: number; xhigh: number; + max: number; } export interface SttSettings { diff --git a/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json b/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json index 69c56126c..80aa3e6cf 100644 --- a/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json +++ b/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json @@ -122,11 +122,12 @@ "status.pending": "◌", "nav.cursor": "▸", "nav.selected": "▴", - "thinking.minimal": "◌", - "thinking.low": "◍", - "thinking.medium": "◎", - "thinking.high": "◉", - "thinking.xhigh": "●", + "thinking.minimal": "∘", + "thinking.low": "◌", + "thinking.medium": "◍", + "thinking.high": "◎", + "thinking.xhigh": "◉", + "thinking.max": "●", "icon.model": "◇", "icon.plan": "◈", "icon.goal": "⊙", diff --git a/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json b/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json index 8b00ca22e..a6da984a9 100644 --- a/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json +++ b/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json @@ -122,11 +122,12 @@ "status.pending": "◌", "nav.cursor": "▸", "nav.selected": "▴", - "thinking.minimal": "◌", - "thinking.low": "◍", - "thinking.medium": "◎", - "thinking.high": "◉", - "thinking.xhigh": "●", + "thinking.minimal": "∘", + "thinking.low": "◌", + "thinking.medium": "◍", + "thinking.high": "◎", + "thinking.xhigh": "◉", + "thinking.max": "●", "icon.model": "◇", "icon.plan": "◈", "icon.goal": "⊙", diff --git a/packages/coding-agent/src/modes/theme/theme-schema.json b/packages/coding-agent/src/modes/theme/theme-schema.json index 3fd9972b4..8dd7eb41c 100644 --- a/packages/coding-agent/src/modes/theme/theme-schema.json +++ b/packages/coding-agent/src/modes/theme/theme-schema.json @@ -33,7 +33,7 @@ }, "colors": { "type": "object", - "description": "Theme color definitions (all required)", + "description": "Theme color definitions (all required except thinkingMax)", "required": [ "accent", "border", @@ -301,7 +301,11 @@ }, "thinkingXhigh": { "$ref": "#/$defs/colorValue", - "description": "Thinking level border: xhigh (OpenAI codex-max only)" + "description": "Thinking level border: xhigh" + }, + "thinkingMax": { + "$ref": "#/$defs/colorValue", + "description": "Thinking level border: max (optional; falls back to thinkingXhigh when omitted)" }, "bashMode": { "$ref": "#/$defs/colorValue", diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index 77140221b..ca0a6397f 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -142,6 +142,7 @@ export type SymbolKey = | "thinking.medium" | "thinking.high" | "thinking.xhigh" + | "thinking.max" | "thinking.autoPending" // Checkboxes | "checkbox.checked" @@ -344,11 +345,12 @@ const UNICODE_SYMBOLS: SymbolMap = { // Compaction divider "icon.camera": "📷", // Thinking levels - "thinking.minimal": "◔ min", - "thinking.low": "◑ low", - "thinking.medium": "◒ med", - "thinking.high": "◕ high", - "thinking.xhigh": "◉ xhigh", + "thinking.minimal": "○ min", + "thinking.low": "◔ low", + "thinking.medium": "◑ med", + "thinking.high": "◒ high", + "thinking.xhigh": "◕ xhigh", + "thinking.max": "◉ max", "thinking.autoPending": "⟳", // Checkboxes "checkbox.checked": "☑", @@ -639,19 +641,15 @@ const NERD_SYMBOLS: SymbolMap = { "icon.mic": "\uf130", // Compaction divider - fa-camera-retro "icon.camera": "\uf083", - // Thinking Levels - emoji labels - // pick: 🤨 min | alt:  min  min - "thinking.minimal": "\u{F0E7} min", - // pick: 🤔 low | alt:  low  low - "thinking.low": "\u{F10C} low", - // pick: 🤓 med | alt:  med  med - "thinking.medium": "\u{F192} med", - // pick: 🤯 high | alt:  high  high - "thinking.high": "\u{F111} high", - // pick: 🧠 xhi | alt:  xhi  xhi - "thinking.xhigh": "\u{F06D} xhi", - // pick: (fa-circle-o-notch) | alt: 󰂼 (nf-md-cached) ⟳ - "thinking.autoPending": "\uf1ce", + // Thinking levels — empty-to-full battery ramp, then fire. + "thinking.minimal": "\u{F244} min", + "thinking.low": "\u{F243} low", + "thinking.medium": "\u{F242} med", + "thinking.high": "\u{F241} high", + "thinking.xhigh": "\u{F240} xhi", + "thinking.max": "\u{F06D} max", + // Auto mode uses shuffle until the model resolves its thinking level. + "thinking.autoPending": "\u{F074}", // Checkboxes // pick:  | alt:   "checkbox.checked": "\uf14a", @@ -868,6 +866,7 @@ const ASCII_SYMBOLS: SymbolMap = { "thinking.medium": "[med]", "thinking.high": "[high]", "thinking.xhigh": "[xhi]", + "thinking.max": "[max]", "thinking.autoPending": "[~]", // Checkboxes "checkbox.checked": "[x]", @@ -1059,6 +1058,7 @@ const themeColorsSchema = type({ thinkingMedium: "string | number", thinkingHigh: "string | number", thinkingXhigh: "string | number", + "thinkingMax?": "string | number", bashMode: "string | number", pythonMode: "string | number", statusLineBg: "string | number", @@ -1163,6 +1163,7 @@ export type ThemeColor = | "thinkingMedium" | "thinkingHigh" | "thinkingXhigh" + | "thinkingMax" | "bashMode" | "pythonMode" | "statusLineSep" @@ -1225,6 +1226,7 @@ const THEME_COLOR_RECORD = { thinkingMedium: true, thinkingHigh: true, thinkingXhigh: true, + thinkingMax: true, bashMode: true, pythonMode: true, statusLineSep: true, @@ -1675,6 +1677,9 @@ export class Theme { return (str: string) => this.fg("thinkingHigh", str); case "xhigh": return (str: string) => this.fg("thinkingXhigh", str); + case "max": + // thinkingMax is optional; themes without it resolve to the xhigh color. + return (str: string) => this.fg(this.#fgColors.thinkingMax ? "thinkingMax" : "thinkingXhigh", str); default: return (str: string) => this.fg("thinkingOff", str); } @@ -1861,6 +1866,7 @@ export class Theme { medium: this.#symbols["thinking.medium"], high: this.#symbols["thinking.high"], xhigh: this.#symbols["thinking.xhigh"], + max: this.#symbols["thinking.max"], autoPending: this.#symbols["thinking.autoPending"], }; } diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 16a066675..8afd72517 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1309,7 +1309,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} for (let i = 0; i < sessionModelStrings.length; i++) { const sessionModelStr = sessionModelStrings[i]; const parsedModel = parseModelString(sessionModelStr, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => modelRegistry.find(provider, id) !== undefined, }); @@ -1953,7 +1953,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} for (let i = 0; i < sessionRetryLimit; i++) { const sessionModelStr = sessionModelStrings[i]; const parsedModel = parseModelString(sessionModelStr, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => modelRegistry.find(provider, id) !== undefined, }); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index a98f48521..e6702ad36 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1081,7 +1081,7 @@ function parseRetryFallbackSelector( const trimmed = selector.trim(); if (!trimmed) return undefined; const parsed = parseModelString(trimmed, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => modelLookup?.find(provider, id) !== undefined, }); @@ -9260,7 +9260,7 @@ export class AgentSession { } /** - * Cycle to next thinking level: off → auto → minimal..xhigh → off. + * Cycle to next thinking level: off → auto → minimal..max → off. * @returns New selector, or undefined if model doesn't support thinking */ cycleThinkingLevel(): ConfiguredThinkingLevel | undefined { @@ -9302,8 +9302,9 @@ export class AgentSession { let resolved: Effort | undefined; if (this.#magicKeywordEnabled("ultrathink") && containsUltrathink(promptText)) { // The user explicitly asked for maximum thinking; bypass the classifier - // and jump straight to the highest auto-supported level for this model. - resolved = clampAutoThinkingEffort(model, Effort.XHigh); + // (and its xhigh auto ceiling) and jump straight to the highest + // supported level for this model. + resolved = clampAutoThinkingEffort(model, Effort.Max); } else { const controller = new AbortController(); const timer = setTimeout(() => controller.abort(), AgentSession.#AUTO_THINKING_TIMEOUT_MS); @@ -11967,7 +11968,7 @@ export class AgentSession { if (!trimmedTarget) return undefined; const parsed = parseModelString(trimmedTarget, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => availableModels.some(model => model.provider === provider && model.id === id), diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index 673085068..2f93dbbc4 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -123,29 +123,36 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { }, 15_000); it("kills the GPU probe at the prep deadline", async () => { - const result = await runProbeScenario({ runs: 1, sleepSeconds: 7, holdStdoutOpen: true }); + const result = await runProbeScenario({ runs: 1, sleepSeconds: 12, holdStdoutOpen: true }); expect(result.cached).toEqual({ gpu: null }); + // Probe is SIGKILLed at ~4.5s and the drain wait is bounded, so in-child + // time sits near the deadline; waiting on the descendant would push it + // past the 12s sleep. expect(result.elapsedMs).toBeLessThan(6500); - // Codex#3838: the child process MUST exit shortly after the deadline, - // not linger until a descendant holding stdout (sleep 7) exits on its own. - expect(result.childElapsedMs).toBeLessThan(6500); - }, 15_000); + // Codex#3838: the child process MUST exit shortly after the deadline, not + // linger until a descendant holding stdout (sleep 12) exits on its own. + // The bound over in-child time budgets bun spawn/startup on loaded runners + // while staying far below the descendant's 12s exit. + expect(result.childElapsedMs).toBeLessThan(9000); + }, 20_000); it("does not wait on stdout held by a descendant after a successful probe", async () => { - const result = await runProbeScenario({ runs: 1, sleepSeconds: 3, descendantHoldsStdout: true }); + const result = await runProbeScenario({ runs: 1, sleepSeconds: 8, descendantHoldsStdout: true }); expect(result.cached).toEqual({ gpu: null }); // Probe exits 0 immediately but leaves a backgrounded sleep holding the stdout // pipe. The success path MUST bound the drain wait, not block until sleep exits. expect(result.elapsedMs).toBeLessThan(2000); - expect(result.childElapsedMs).toBeLessThan(2000); - }, 15_000); + // Budgets bun spawn/startup overhead; blocking on the descendant would + // take at least the 8s sleep. + expect(result.childElapsedMs).toBeLessThan(5000); + }, 20_000); it("keeps probe output captured before a descendant delays EOF", async () => { const result = await runProbeScenario({ runs: 1, - sleepSeconds: 3, + sleepSeconds: 8, descendantHoldsStdout: true, validOutput: "00:02.0 VGA compatible controller: NVIDIA TestGPU", }); @@ -154,8 +161,10 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { // Captured stdout MUST be cached, not discarded as if the probe failed. expect(result.cached).toEqual({ gpu: "02.0 VGA compatible controller: NVIDIA TestGPU" }); expect(result.elapsedMs).toBeLessThan(2000); - expect(result.childElapsedMs).toBeLessThan(2000); - }, 15_000); + // Budgets bun spawn/startup overhead; blocking on the descendant would + // take at least the 8s sleep. + expect(result.childElapsedMs).toBeLessThan(5000); + }, 20_000); }); describe.skipIf(process.platform !== "linux")("system prompt CPU model", () => { diff --git a/packages/coding-agent/src/thinking.ts b/packages/coding-agent/src/thinking.ts index 7a6e83a8b..0bf10760c 100644 --- a/packages/coding-agent/src/thinking.ts +++ b/packages/coding-agent/src/thinking.ts @@ -33,7 +33,12 @@ const THINKING_LEVEL_METADATA: Record = { [ThinkingLevel.XHigh]: { value: ThinkingLevel.XHigh, label: "xhigh", - description: "Maximum reasoning (~32k tokens)", + description: "Extended reasoning (~32k tokens)", + }, + [ThinkingLevel.Max]: { + value: ThinkingLevel.Max, + label: "max", + description: "Maximum reasoning the model supports", }, }; @@ -43,7 +48,7 @@ const EFFORT_BY_SELECTOR: Readonly> = { [Effort.Medium]: Effort.Medium, [Effort.High]: Effort.High, [Effort.XHigh]: Effort.XHigh, - max: Effort.XHigh, + [Effort.Max]: Effort.Max, }; const THINKING_LEVEL_BY_SELECTOR: Readonly> = { [ThinkingLevel.Inherit]: ThinkingLevel.Inherit, @@ -53,6 +58,7 @@ const THINKING_LEVEL_BY_SELECTOR: Readonly> = { [ThinkingLevel.Medium]: ThinkingLevel.Medium, [ThinkingLevel.High]: ThinkingLevel.High, [ThinkingLevel.XHigh]: ThinkingLevel.XHigh, + [ThinkingLevel.Max]: ThinkingLevel.Max, }; function getOwnSelector(selectors: Readonly>, value: string | null | undefined): T | undefined { @@ -149,7 +155,6 @@ const AUTO_THINKING_METADATA: ConfiguredThinkingLevelMetadata = { */ export function parseConfiguredThinkingLevel(value: string | null | undefined): ConfiguredThinkingLevel | undefined { if (value === AUTO_THINKING) return AUTO_THINKING; - if (value === "max") return ThinkingLevel.XHigh; return parseThinkingLevel(value); } @@ -160,7 +165,7 @@ export function getConfiguredThinkingLevelMetadata(level: ConfiguredThinkingLeve /** * Thinking selectors accepted by the `--thinking` CLI flag, in display order: - * `off`, every concrete effort (`minimal`..`xhigh`), then `auto`. Single source + * `off`, every concrete effort (`minimal`..`max`), then `auto`. Single source * for the flag's `options` list, shell completions, and the "invalid level" * warning so all three stay in sync. */ @@ -168,7 +173,7 @@ export const CLI_THINKING_LEVELS: readonly string[] = [ThinkingLevel.Off, ...THI /** * Parses a `--thinking` CLI value. Accepts every {@link parseConfiguredThinkingLevel} - * selector (`off`, `auto`, `minimal`..`xhigh`, plus the `max` alias) but rejects + * selector (`off`, `auto`, `minimal`..`max`) but rejects * `inherit`: an explicit `inherit` on the command line would suppress the * settings/scoped-model fallback during startup resolution only to resolve back * to the provider default, which is never what the user means. @@ -211,9 +216,13 @@ export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort /** * The provisional concrete level shown while `auto` is configured but before a * turn has been classified. Prefers the model's `defaultLevel`, otherwise High, - * clamped into the auto range. Returns `undefined` for non-reasoning models. + * clamped into the auto range. Auto never provisions {@link Effort.Max} (the + * classifier ceiling is XHigh; only an explicit user request reaches Max), so a + * `defaultLevel` of `max` is capped at XHigh before clamping. Returns + * `undefined` for non-reasoning models. */ export function resolveProvisionalAutoLevel(model: Model | undefined): Effort | undefined { if (!model?.reasoning) return undefined; - return clampAutoThinkingEffort(model, model.thinking?.defaultLevel ?? Effort.High); + const preferred = model.thinking?.defaultLevel ?? Effort.High; + return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred); } diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 345d42df3..25c8c6791 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -8,7 +8,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; -import { parseModelPattern } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { parseModelPattern, parseModelString } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -1602,6 +1602,12 @@ describe("AgentSession retry fallback", () => { expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o")).toBe(true); expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o:low")).toBe(true); + // `:max` is a real thinking level now, not an xhigh alias — the two parse + // to distinct selectors... + expect(parseModelString("openai/gpt-4o:max", { allowMaxSuffix: true })?.thinkingLevel).toBe(Effort.Max); + expect(parseModelString("openai/gpt-4o:xhigh")?.thinkingLevel).toBe(Effort.XHigh); + // ...but suppression normalizes every thinking suffix to the base selector, + // so suppressing either still covers both. modelRegistry.suppressSelector("openai/gpt-4o:max", future); expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o:xhigh")).toBe(true); expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o:max")).toBe(true); diff --git a/packages/coding-agent/test/agent-session-role-thinking.test.ts b/packages/coding-agent/test/agent-session-role-thinking.test.ts index a489a02b6..910908325 100644 --- a/packages/coding-agent/test/agent-session-role-thinking.test.ts +++ b/packages/coding-agent/test/agent-session-role-thinking.test.ts @@ -147,14 +147,16 @@ describe("AgentSession role model thinking behavior", () => { expect(toSlow?.thinkingLevel).toBe(Effort.High); expect(session.thinkingLevel).toBe(Effort.High); - session.setThinkingLevel(Effort.Minimal); - expect(session.thinkingLevel).toBe(Effort.Minimal); + // `medium` is supported on both ladders (4-6 dropped `minimal`), so the + // selection survives the role switch unclamped. + session.setThinkingLevel(Effort.Medium); + expect(session.thinkingLevel).toBe(Effort.Medium); const toDefault = await session.cycleRoleModels(["default", "slow"]); expect(toDefault?.role).toBe("default"); expect(toDefault?.model.id).toBe(defaultModel.id); - expect(toDefault?.thinkingLevel).toBe(Effort.Minimal); - expect(session.thinkingLevel).toBe(Effort.Minimal); + expect(toDefault?.thinkingLevel).toBe(Effort.Medium); + expect(session.thinkingLevel).toBe(Effort.Medium); }); it("applies slow role thinking even when plan shares the same model", async () => { @@ -231,6 +233,36 @@ describe("AgentSession role model thinking behavior", () => { expect(session.getAvailableThinkingLevels()).not.toContain("xhigh"); }); + it("clamps max selections down to the ladder ceiling on models without a max tier", async () => { + // Budget-mode sonnet-4-5 tops out at xhigh; a max request must clamp down. + const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); + const agent = new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + thinkingLevel: undefined, + }, + }); + const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-max-clamp.db")); + authStorages.push(authStorage); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-max-clamp.yml")); + + sessionSettings = Settings.isolated(); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: sessionSettings, + modelRegistry, + }); + + session.setThinkingLevel(Effort.Max); + expect(session.thinkingLevel).toBe(Effort.XHigh); + expect(session.getAvailableThinkingLevels()).not.toContain("max"); + }); + it("cycles through off and auto before returning to effort levels", async () => { const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); @@ -267,6 +299,40 @@ describe("AgentSession role model thinking behavior", () => { expect(session.thinkingLevel).toBe(Effort.Minimal); }); + it("cycles through max as the final tier on a max-capable model", async () => { + const model = getAnthropicModelOrThrow("claude-opus-4-7"); + const agent = new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + thinkingLevel: Effort.XHigh, + }, + }); + const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-cycle-max.db")); + authStorages.push(authStorage); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-cycle-max.yml")); + + sessionSettings = Settings.isolated(); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: sessionSettings, + modelRegistry, + }); + + const available = session.getAvailableThinkingLevels(); + expect(available.at(-1)).toBe(Effort.Max); + + session.setThinkingLevel(Effort.XHigh); + expect(session.cycleThinkingLevel()).toBe(Effort.Max); + expect(session.thinkingLevel).toBe(Effort.Max); + // max is the last tier: the wheel wraps back to off. + expect(session.cycleThinkingLevel()).toBe("off"); + }); + it("keeps auto configured while applying the classifier result as the effective level", async () => { const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); await createSession({ @@ -472,7 +538,7 @@ describe("AgentSession role model thinking behavior", () => { expect(session.autoResolvedThinkingLevel()).toBeUndefined(); }); - it("maps ultrathink prompts directly to the highest auto-supported level", async () => { + it("maps ultrathink prompts to the model's highest supported level, clamped below max", async () => { const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); await createSession({ initialModelId: model.id, @@ -483,7 +549,9 @@ describe("AgentSession role model thinking behavior", () => { const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Low); session.setThinkingLevel(AUTO_THINKING); - const expected = clampAutoThinkingEffort(model, Effort.XHigh); + // sonnet-4-5 has no max tier, so the ultrathink jump clamps to xhigh. + const expected = clampAutoThinkingEffort(model, Effort.Max); + expect(expected).toBe(Effort.XHigh); await session.prompt("ultrathink through the unsafe refactor"); expect(classifierSpy).not.toHaveBeenCalled(); @@ -491,6 +559,24 @@ describe("AgentSession role model thinking behavior", () => { expect(session.autoResolvedThinkingLevel()).toBe(expected); }); + it("resolves ultrathink to max on max-capable models", async () => { + const model = getAnthropicModelOrThrow("claude-opus-4-7"); + await createSession({ + initialModelId: model.id, + initialThinkingLevel: Effort.High, + modelRoles: { default: `${model.provider}/${model.id}` }, + }); + vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Low); + + session.setThinkingLevel(AUTO_THINKING); + await session.prompt("ultrathink through the unsafe refactor"); + + expect(classifierSpy).not.toHaveBeenCalled(); + expect(session.thinkingLevel).toBe(Effort.Max); + expect(session.autoResolvedThinkingLevel()).toBe(Effort.Max); + }); + it("keeps auto effectively off for non-reasoning models", async () => { const model = getBundledModel("openai", "gpt-4o-mini"); if (!model) throw new Error("Expected bundled gpt-4o-mini model"); diff --git a/packages/coding-agent/test/auto-thinking-classifier.test.ts b/packages/coding-agent/test/auto-thinking-classifier.test.ts index 912b1e6d8..29f67b392 100644 --- a/packages/coding-agent/test/auto-thinking-classifier.test.ts +++ b/packages/coding-agent/test/auto-thinking-classifier.test.ts @@ -3,6 +3,7 @@ import * as path from "node:path"; import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import * as ai from "@oh-my-pi/pi-ai"; import { Effort, type Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { classifyDifficulty, @@ -71,7 +72,7 @@ describe("auto thinking classifier helpers", () => { it("parses CLI --thinking selectors while rejecting inherit", () => { expect(parseCliThinkingLevel(ThinkingLevel.Off)).toBe(ThinkingLevel.Off); expect(parseCliThinkingLevel(AUTO_THINKING)).toBe(AUTO_THINKING); - expect(parseCliThinkingLevel("max")).toBe(ThinkingLevel.XHigh); + expect(parseCliThinkingLevel("max")).toBe(ThinkingLevel.Max); expect(parseCliThinkingLevel(ThinkingLevel.Inherit)).toBeUndefined(); expect(parseCliThinkingLevel("bogus")).toBeUndefined(); }); @@ -182,6 +183,24 @@ describe("auto thinking classifier helpers", () => { expect(clampAutoThinkingEffort(model, Effort.Minimal)).toBe(Effort.Low); }); + it("clamps max down to the ladder ceiling on models without a max tier", () => { + const xhighCeilingModel = buildModel({ + id: "mock-xhigh-ceiling", + name: "Mock XHigh Ceiling", + api: "openai-completions", + provider: "mock", + baseUrl: "https://example.com", + reasoning: true, + thinking: { mode: "effort", efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 4096, + }); + + expect(clampAutoThinkingEffort(xhighCeilingModel, Effort.Max)).toBe(Effort.XHigh); + }); + it("returns undefined for reasoning models without controllable efforts (devin-agent shape)", () => { // Repro for https://github.com/can1357/oh-my-pi/issues/3356 — Devin // models report `reasoning: true` but expose no `thinking.efforts` (Cascade @@ -202,13 +221,14 @@ describe("auto thinking classifier helpers", () => { expect(clampAutoThinkingEffort(devinModel, Effort.Low)).toBeUndefined(); expect(clampAutoThinkingEffort(devinModel, Effort.XHigh)).toBeUndefined(); + expect(clampAutoThinkingEffort(devinModel, Effort.Max)).toBeUndefined(); expect(resolveProvisionalAutoLevel(devinModel)).toBeUndefined(); }); - it("accepts max as the top configured thinking alias", () => { - expect(parseEffort("max")).toBe(Effort.XHigh); - expect(parseThinkingLevel("max")).toBeUndefined(); - expect(parseConfiguredThinkingLevel("max")).toBe(ThinkingLevel.XHigh); + it("parses max as a real thinking level", () => { + expect(parseEffort("max")).toBe(Effort.Max); + expect(parseThinkingLevel("max")).toBe(ThinkingLevel.Max); + expect(parseConfiguredThinkingLevel("max")).toBe(ThinkingLevel.Max); }); it("rejects inherited object keys as thinking selectors", () => { diff --git a/packages/coding-agent/test/cli-hide-thinking-flag.test.ts b/packages/coding-agent/test/cli-hide-thinking-flag.test.ts index 40c4d548b..807acd653 100644 --- a/packages/coding-agent/test/cli-hide-thinking-flag.test.ts +++ b/packages/coding-agent/test/cli-hide-thinking-flag.test.ts @@ -52,10 +52,10 @@ describe("parseArgs — --thinking flag", () => { expect(parseArgs(["--thinking=off"]).thinking).toBe(ThinkingLevel.Off); }); - it("accepts auto, concrete efforts, and the max alias", () => { + it("accepts auto and every concrete effort including max", () => { expect(parseArgs(["--thinking", "auto"]).thinking).toBe(AUTO_THINKING); expect(parseArgs(["--thinking", "medium"]).thinking).toBe(Effort.Medium); - expect(parseArgs(["--thinking", "max"]).thinking).toBe(ThinkingLevel.XHigh); + expect(parseArgs(["--thinking", "max"]).thinking).toBe(ThinkingLevel.Max); }); it("ignores invalid levels and the internal inherit selector", () => { diff --git a/packages/coding-agent/test/cli/completions.test.ts b/packages/coding-agent/test/cli/completions.test.ts index 4da7f19ef..3dd8dd9a6 100644 --- a/packages/coding-agent/test/cli/completions.test.ts +++ b/packages/coding-agent/test/cli/completions.test.ts @@ -211,7 +211,7 @@ describe("omp completions (integration / drift)", () => { } expect(stdout).toContain("{-r,--resume}"); // Real enum option sets flow through unchanged. - expect(stdout).toContain(":value:(off minimal low medium high xhigh auto)"); + expect(stdout).toContain(":value:(off minimal low medium high xhigh max auto)"); expect(stdout).toContain(":value:(always-ask write yolo)"); // Real subcommands present; dynamic callbacks wired. expect(stdout).toContain("_omp_cmd_commit"); diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index e3e36fd96..4e2e5d4fc 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -219,6 +219,25 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [ }), ]; +const mockMaxCapableModels: Model<"anthropic-messages">[] = [ + buildModel({ + id: "claude-opus-4-7", + name: "Claude Opus 4.7", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + }, + input: ["text", "image"], + cost: { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 }, + contextWindow: 200000, + maxTokens: 32000, + }), +]; + const openaiGpt55Models: Model[] = [ buildModel({ id: "gpt-5.5", @@ -410,7 +429,15 @@ describe("parseModelPattern", () => { }); test("all valid thinking levels work", () => { - const levels = ["off", Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] as const; + const levels = [ + "off", + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + Effort.Max, + ] as const; for (const level of levels) { const result = parseModelPattern(`sonnet:${level}`, allModels); expect(result.model?.id).toBe("claude-sonnet-4-5"); @@ -418,15 +445,15 @@ describe("parseModelPattern", () => { expect(result.warning).toBeUndefined(); } }); - test("max aliases the highest thinking level after the literal pattern misses", () => { + test("max parses as a real thinking level after the literal pattern misses", () => { const result = parseModelPattern("gpt-5.3-codex:max", allModels); expect(result.model?.id).toBe("gpt-5.3-codex"); - expect(result.thinkingLevel).toBe(Effort.XHigh); + expect(result.thinkingLevel).toBe(Effort.Max); expect(result.explicitThinkingLevel).toBe(true); expect(result.warning).toBeUndefined(); }); - test("literal model ids ending in max win over the thinking alias", () => { + test("literal model ids ending in max win over the thinking suffix", () => { const result = parseModelPattern("nanogpt/coding-router:max", mockMaxSuffixModels); expect(result.model?.id).toBe("coding-router:max"); expect(result.thinkingLevel).toBeUndefined(); @@ -523,13 +550,13 @@ describe("parseModelPattern", () => { expect(result.warning).toBeUndefined(); }); - test("openrouter/:max applies xhigh through the exact-selector path, not an OpenRouter route", () => { - // `max` is a thinking alias, never an OpenRouter route suffix: the request must - // resolve the base model and carry xhigh, not clone a literal `z-ai/glm-4.7:max`. + test("openrouter/:max applies max through the exact-selector path, not an OpenRouter route", () => { + // `max` is a thinking-level suffix, never an OpenRouter route suffix: the request + // must resolve the base model and carry max, not clone a literal `z-ai/glm-4.7:max`. const result = parseModelPattern("openrouter/z-ai/glm-4.7:max", allModels); expect(result.model?.provider).toBe("openrouter"); expect(result.model?.id).toBe("z-ai/glm-4.7"); - expect(result.thinkingLevel).toBe(Effort.XHigh); + expect(result.thinkingLevel).toBe(Effort.Max); expect(result.explicitThinkingLevel).toBe(true); }); }); @@ -618,6 +645,7 @@ describe("resolveModelRoleValue", () => { expect(result.model?.provider).toBe("openai-codex"); expect(result.model?.id).toBe("gpt-5.3-codex"); + // Role-value resolution clamps: gpt-5.3-codex's ladder tops out at xhigh. expect(result.thinkingLevel).toBe(Effort.XHigh); expect(result.explicitThinkingLevel).toBe(true); }); @@ -679,6 +707,15 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); + test("passes max through unclamped when the model ladder includes it", () => { + const result = resolveModelRoleValue("anthropic/claude-opus-4-7:max", mockMaxCapableModels); + + expect(result.model?.provider).toBe("anthropic"); + expect(result.model?.id).toBe("claude-opus-4-7"); + expect(result.thinkingLevel).toBe(Effort.Max); + expect(result.explicitThinkingLevel).toBe(true); + }); + test("preserves an explicit :auto suffix as an explicit thinking selector", () => { const result = resolveModelRoleValue("anthropic/claude-sonnet-4-5:auto", allModels); @@ -1071,7 +1108,7 @@ describe("resolveModelScope", () => { expect(scoped[0].model.id).toBe("gpt-5.5"); }); - test("applies max thinking aliases to glob scopes when no literal max ids match", async () => { + test("applies max thinking selectors to glob scopes when no literal max ids match", async () => { const registry = { getAvailable: () => mockCodexOverlapModels, }; @@ -1079,10 +1116,23 @@ describe("resolveModelScope", () => { const scoped = await resolveModelScope(["openai-codex/*:max"], registry); expect(scoped).toHaveLength(2); + // Scoped levels clamp per model: max on an xhigh-ceiling ladder resolves to xhigh. expect(scoped.map(entry => entry.thinkingLevel)).toEqual([Effort.XHigh, Effort.XHigh]); expect(scoped.every(entry => entry.explicitThinkingLevel)).toBe(true); }); + test("keeps max on glob scopes when the model ladder includes it", async () => { + const registry = { + getAvailable: () => mockMaxCapableModels, + }; + + const scoped = await resolveModelScope(["anthropic/*:max"], registry); + + expect(scoped).toHaveLength(1); + expect(scoped[0].thinkingLevel).toBe(Effort.Max); + expect(scoped[0].explicitThinkingLevel).toBe(true); + }); + test("preserves literal :max in scoped-model globs", async () => { const registry = { getAvailable: () => mockMaxSuffixModels, @@ -1144,18 +1194,25 @@ describe("parseModelString", () => { }); test("extracts max when explicitly enabled for provider id selectors", () => { - const result = parseModelString("deepseek/deepseek-v4-pro:max", { allowMaxAlias: true }); - expect(result).toEqual({ provider: "deepseek", id: "deepseek-v4-pro", thinkingLevel: Effort.XHigh }); + const result = parseModelString("deepseek/deepseek-v4-pro:max", { allowMaxSuffix: true }); + expect(result).toEqual({ provider: "deepseek", id: "deepseek-v4-pro", thinkingLevel: Effort.Max }); }); test("preserves literal max model ids when the caller can prove they exist", () => { const result = parseModelString("nanogpt/coding-router:max", { - allowMaxAlias: true, + allowMaxSuffix: true, isLiteralModelId: (provider, id) => provider === "nanogpt" && id === "coding-router:max", }); expect(result).toEqual({ provider: "nanogpt", id: "coding-router:max" }); }); + test("leaves :max attached to the model id unless the caller opts in via allowMaxSuffix", () => { + // Without allowMaxSuffix, the strict suffix parser must not silently + // reinterpret a literal `:max` id as a thinking suffix. + const result = parseModelString("anthropic/claude-sonnet-4-5:max"); + expect(result).toEqual({ provider: "anthropic", id: "claude-sonnet-4-5:max" }); + }); + test("leaves :auto attached to the model id unless the caller opts in via allowAutoAlias", () => { // Without allowAutoAlias, the strict suffix parser must not silently // reinterpret a literal `:auto` id as an auto-thinking selector. @@ -1242,7 +1299,7 @@ describe("extractExplicitThinkingSelector", () => { const result = extractExplicitThinkingSelector("nanogpt/coding-router:max", undefined, { isLiteralModelId: () => false, }); - expect(result).toBe(Effort.XHigh); + expect(result).toBe(Effort.Max); }); test("treats max on pi role aliases as an explicit selector before expansion", () => { @@ -1251,7 +1308,7 @@ describe("extractExplicitThinkingSelector", () => { const result = extractExplicitThinkingSelector("pi/smol:max", settings, { isLiteralModelId: (provider, id) => provider === "nanogpt" && id === "coding-router:max", }); - expect(result).toBe(Effort.XHigh); + expect(result).toBe(Effort.Max); }); test("does not carry auto from literal role model ids", () => { diff --git a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts index 50db05d61..0b9da2aec 100644 --- a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts +++ b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts @@ -171,7 +171,7 @@ describe("ModelSelector role badge thinking display", () => { expect(menuRendered).toContain("Set as SMOL (Quick)"); }); - test("renders xhigh effort for OpenAI GPT-5.5 thinking options", async () => { + test("renders the xhigh-ceiling ladder without a speculative max tier (GPT-5.5)", async () => { installTestTheme(); const model = getBundledModel("openai", "gpt-5.5"); if (!model) throw new Error("Expected bundled model openai/gpt-5.5"); @@ -186,7 +186,25 @@ describe("ModelSelector role badge thinking display", () => { const rendered = normalizeRenderedText(selector.render(220).join("\n")); expect(rendered).toContain("Thinking for: Default (gpt-5.5)"); expect(rendered).toContain("low medium high xhigh"); - expect(rendered).not.toContain("low medium high max"); + // gpt-5.5's wire has no max tier; the selector must not invent one. + expect(rendered).not.toContain("max"); + }); + + test("renders max as a real final tier on max-capable models (GPT-5.6)", async () => { + installTestTheme(); + const model = getBundledModel("openai", "gpt-5.6"); + if (!model) throw new Error("Expected bundled model openai/gpt-5.6"); + + const selector = createSelector(model, Settings.isolated({})); + await Bun.sleep(0); + installTestTheme(); + + selector.handleInput("\n"); + selector.handleInput("\n"); + + const rendered = normalizeRenderedText(selector.render(220).join("\n")); + expect(rendered).toContain("Thinking for: Default (gpt-5.6)"); + expect(rendered).toContain("low medium high xhigh max"); }); test("reloads DEFAULT(auto) from defaultThinkingLevel", async () => { diff --git a/packages/coding-agent/test/sdk-model-selection.test.ts b/packages/coding-agent/test/sdk-model-selection.test.ts index ab3c0396c..e2706d95e 100644 --- a/packages/coding-agent/test/sdk-model-selection.test.ts +++ b/packages/coding-agent/test/sdk-model-selection.test.ts @@ -277,7 +277,7 @@ describe("createAgentSession deferred model pattern resolution", () => { expect(session.thinkingLevel).toBe("off"); }); - test("normalizes max default thinking level from settings", async () => { + test("clamps a max default thinking level to the model's ladder ceiling", async () => { const settings = Settings.isolated({ defaultThinkingLevel: "max" }); const { session } = await createAgentSession({ @@ -287,6 +287,8 @@ describe("createAgentSession deferred model pattern resolution", () => { expect(session.model?.provider).toBe("runtime-provider"); expect(session.model?.id).toBe("runtime-reasoning-model"); + // The extension model has no explicit ladder; the inferred fallback tops + // out at xhigh, so the real max level clamps down. expect(session.thinkingLevel).toBe(Effort.XHigh); }); diff --git a/packages/terminal-bench/README.md b/packages/terminal-bench/README.md index 1c5786942..4dd12f551 100644 --- a/packages/terminal-bench/README.md +++ b/packages/terminal-bench/README.md @@ -65,7 +65,7 @@ bun src/runner.ts [options] [-- ] | `-n, --concurrency ` | `4` | Concurrent trials | | `-k, --attempts ` | `1` | Attempts per task (pass@k) | | `-i/-x, --include/--exclude ` | — | Task filters (repeatable) | -| `--thinking ` | — | `off…xhigh` | +| `--thinking ` | — | `off…max` | | `--advisor-model

` | — | Second model reviewing the primary; spend summed in | | `--agent ` | `omp` | `oracle`/`nop`/any harbor agent (bypasses omp) | | `--install ` | `local` | `published` = npm `@oh-my-pi/pi-coding-agent` | diff --git a/packages/terminal-bench/src/runner.ts b/packages/terminal-bench/src/runner.ts index 33b24c52e..2aa0ee02c 100755 --- a/packages/terminal-bench/src/runner.ts +++ b/packages/terminal-bench/src/runner.ts @@ -113,7 +113,7 @@ Model / agent: --agent omp (default) | oracle | nop | any harbor agent --install omp source. local = pack /work/pi (default) --version omp version for published install (default: latest) - --thinking off|minimal|low|medium|high|xhigh + --thinking off|minimal|low|medium|high|xhigh|max --advisor-model

Second model reviewing the primary (spend summed in) --advisor-sync Advisor catch-up backlog (default 1 = accurate spend; off = faster) --tarball Reuse a prebuilt omp tarball (implies --no-build) diff --git a/packages/typescript-edit-benchmark/src/index.ts b/packages/typescript-edit-benchmark/src/index.ts index fcb0035da..aebfa8e0a 100755 --- a/packages/typescript-edit-benchmark/src/index.ts +++ b/packages/typescript-edit-benchmark/src/index.ts @@ -116,7 +116,7 @@ Usage: Options: --model Provider/model ID, e.g. anthropic/claude-sonnet-4-20250514 (default) --provider Override provider (auto-detected from model prefix if omitted) - --thinking Thinking level: off, minimal, low, medium, high, xhigh + --thinking Thinking level: off, minimal, low, medium, high, xhigh, max --runs Runs per task (default: 1) --timeout Timeout per run in ms (default: 120000) --connection-timeout Timeout for first event before fast-retry (default: 30000) diff --git a/python/omp-rpc/src/omp_rpc/protocol.py b/python/omp-rpc/src/omp_rpc/protocol.py index a761e4d34..ccd94cee7 100644 --- a/python/omp-rpc/src/omp_rpc/protocol.py +++ b/python/omp-rpc/src/omp_rpc/protocol.py @@ -11,8 +11,10 @@ JsonValue: TypeAlias = JsonPrimitive | list["JsonValue"] | dict[str, "JsonValue" JsonObject: TypeAlias = dict[str, JsonValue] Attribution: TypeAlias = Literal["user", "agent"] -Effort: TypeAlias = Literal["minimal", "low", "medium", "high", "xhigh"] -ThinkingLevel: TypeAlias = Literal["off", "minimal", "low", "medium", "high", "xhigh"] +Effort: TypeAlias = Literal["minimal", "low", "medium", "high", "xhigh", "max"] +ThinkingLevel: TypeAlias = Literal[ + "off", "minimal", "low", "medium", "high", "xhigh", "max" +] StreamingBehavior: TypeAlias = Literal["steer", "followUp"] SteeringMode: TypeAlias = Literal["all", "one-at-a-time"] InterruptMode: TypeAlias = Literal["immediate", "wait"] @@ -50,7 +52,7 @@ VALUE_EXTENSION_UI_METHODS: Final[frozenset[ValueExtensionUiMethod]] = frozenset {"select", "input", "editor"} ) _EFFORT_VALUES: Final[frozenset[str]] = frozenset( - {"minimal", "low", "medium", "high", "xhigh"} + {"minimal", "low", "medium", "high", "xhigh", "max"} ) _THINKING_LEVEL_VALUES: Final[frozenset[str]] = _EFFORT_VALUES | frozenset({"off"}) _STEERING_MODE_VALUES: Final[frozenset[str]] = frozenset({"all", "one-at-a-time"}) diff --git a/python/omp-rpc/uv.lock b/python/omp-rpc/uv.lock new file mode 100644 index 000000000..3966be0f4 --- /dev/null +++ b/python/omp-rpc/uv.lock @@ -0,0 +1,8 @@ +version = 1 +revision = 3 +requires-python = ">=3.11" + +[[package]] +name = "omp-rpc" +version = "0.1.0" +source = { editable = "." } diff --git a/python/robomp/src/config.py b/python/robomp/src/config.py index 6b645f90c..af6abc285 100644 --- a/python/robomp/src/config.py +++ b/python/robomp/src/config.py @@ -10,7 +10,7 @@ from typing import Literal from pydantic import Field, SecretStr, field_validator, model_validator from pydantic_settings import BaseSettings, SettingsConfigDict -ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh"] +ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh", "max"] class Settings(BaseSettings): diff --git a/python/robomp/src/pragmas.py b/python/robomp/src/pragmas.py index 1f5a9cd92..4f4114aa7 100644 --- a/python/robomp/src/pragmas.py +++ b/python/robomp/src/pragmas.py @@ -33,7 +33,7 @@ Supported keys (today): contains `` (case-insensitive). Falls back to the normal random pool selection if no member matches. - `/thinking ` — override `ROBOMP_THINKING` for this run. Accepts - `off|none|no`, `lo|low`, `med|medium`, `hi|high`, `xhi|xhigh` + `off|none|no`, `lo|low`, `med|medium`, `hi|high`, `xhi|xhigh`, `max` (case-insensitive); anything else is ignored. Parser semantics: @@ -49,7 +49,7 @@ from __future__ import annotations import re from typing import Literal -ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh"] +ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh", "max"] # Key = ascii lowercase / digit / dash / underscore, must start with a letter. # The value (when using `/key=value` form) runs to end-of-token. @@ -164,6 +164,7 @@ _THINKING_ALIASES: dict[str, ThinkingLevel] = { "high": "high", "xhi": "xhigh", "xhigh": "xhigh", + "max": "max", } diff --git a/scripts/edit_benchmark_common.py b/scripts/edit_benchmark_common.py index 2b7c37859..3280fc1a8 100644 --- a/scripts/edit_benchmark_common.py +++ b/scripts/edit_benchmark_common.py @@ -654,7 +654,7 @@ def install_verbose_logging( with _PRINT_LOCK: sys.stderr.write( f"[{model.removeprefix('openrouter/')}] verbose> " - "no thinking level requested; pass --thinking low|medium|high|xhigh if the provider exposes reasoning.\n" + "no thinking level requested; pass --thinking low|medium|high|xhigh|max if the provider exposes reasoning.\n" ) sys.stderr.flush() @@ -914,7 +914,7 @@ def parse_args(description: str) -> argparse.Namespace: ) parser.add_argument( "--thinking", - choices=["off", "minimal", "low", "medium", "high", "xhigh"], + choices=["off", "minimal", "low", "medium", "high", "xhigh", "max"], default="medium", help="Request a specific thinking level for models that support reasoning (default: medium).", ) From eae4580a36c344be93a4fa718bdaa2b17f8b1ef9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 13:40:13 +0200 Subject: [PATCH 067/205] chore: update changelogs --- packages/agent/CHANGELOG.md | 6 ++--- packages/ai/CHANGELOG.md | 31 ++++++++---------------- packages/catalog/CHANGELOG.md | 24 ++++++++---------- packages/coding-agent/CHANGELOG.md | 39 ++++++++++++------------------ packages/tui/CHANGELOG.md | 2 +- 5 files changed, 39 insertions(+), 63 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index c8702a17d..b5ebcd148 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -4,12 +4,12 @@ ### Added -- Added `ThinkingLevel.Max` ("max") above `xhigh`, mapping to the catalog `Effort.Max` tier. +- Added the `ThinkingLevel.Max` ("max") configuration option, mapping to the `Effort.Max` tier for supported models. ### Fixed -- Fixed remote compaction for Codex Responses Lite models (GPT-5.6 family): both the V1 `/responses/compact` request and the V2 `compaction_trigger` stream now apply the lite rewrite (instructions as an input item, no top-level `instructions`/`tools`, `all_turns` reasoning replay on V2) and send the `x-openai-internal-codex-responses-lite` header, matching codex-rs routing compaction through `build_responses_request`. -- Fixed aborted tool-result hooks from continuing into another provider call before the abort settled. ([#4963](https://github.com/can1357/oh-my-pi/issues/4963)) +- Fixed remote compaction behavior for Codex Responses Lite (GPT-5.6 family) models across both V1 and V2 endpoints to ensure correct formatting and routing. +- Fixed an issue where aborted tool-result hooks could trigger subsequent provider calls before the abort signal fully settled. ## [16.3.12] - 2026-07-08 diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index b2c3669e1..7478d451b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,33 +4,22 @@ ### Added -- Added `max` as a first-class reasoning effort option across providers and wire schemas -- Updated `max` reasoning budget to 32768 tokens across all provider budget tables - -- Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. -- Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`. -- Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress. -- Added Novita API-key login with authenticated key validation and `NOVITA_API_KEY` discovery ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). -- Added `"max"` as a first-class reasoning effort across provider options, wire types (`reasoning_effort`/`reasoning.effort`), server intake guards, and the Codex request transformer; user `max` now serializes 1:1 to the provider `max` tier instead of being reachable only through the retired shifted effort maps. Inbound Anthropic gateway requests now map `output_config.effort` onto `options.reasoning`. +- Added "max" as a first-class reasoning effort option across providers (including Anthropic, Google, Bedrock, and OpenAI), supporting a maximum reasoning budget of 32,768 tokens. +- Added and standardized the "Responses Lite" wire contract and transport, enabling automatic activation via model-level catalog flags, moving tools and instructions into developer input items, disabling parallel tool calls, and stripping image detail instead of falling back to the full transport. +- Added support for concurrent reasoning summaries on Codex Responses using the sequential-cutoff streaming contract. +- Added Novita API-key login with authenticated key validation and automatic NOVITA_API_KEY environment variable discovery. ### Changed -- Refactored Responses Lite transport to move tools and instructions into input items -- Updated Responses Lite to force parallel tool calling off and strip image detail -- Standardized Responses Lite activation via model-level catalog flags -- Recognized Pro Lite as a paid plan tier for OpenAI Codex models -- Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape. -- Changed effort budget tables (`ANTHROPIC_THINKING`, `GOOGLE_THINKING`, `BEDROCK_CLAUDE_THINKING`, Bedrock `defaultBudgets`) to carry a `max` row (32768), and `getGoogleBudget` to resolve `max` to the largest bucket explicitly. +- Recognized Pro Lite as a paid plan tier for OpenAI Codex models. ### Fixed -- Fixed xAI SuperGrok multi-account rotation when an account returns HTTP 403 `run out of credits` / `personal-team-blocked:spending-limit`. That account-local cap is now classified as a usage limit so `streamSimple` auth-retry and `rotateSessionCredential` switch to a sibling `xai-oauth` credential instead of sticking to the exhausted account. -- Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract -- Fixed sequential-cutoff Codex reasoning summaries repeating earlier content when atomic summary snapshots are replayed or extended. -- Fixed error classification for typed AWS credential-resolution failures (`AwsCredentialsError`) to map them to authentication failures. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) -- Fixed OpenAI-compatible chat-completions streams preserving vLLM-style trailing cached-token usage chunks so `cacheRead` and billable `input` session stats are accurate ([#5022](https://github.com/can1357/oh-my-pi/issues/5022)). -- Fixed `xai-oauth/grok-4.5` Responses requests to omit unsupported `reasoning.summary` while preserving the documented `reasoning.effort` payload ([#4998](https://github.com/can1357/oh-my-pi/issues/4998)). -- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks when live usage shows all reported windows recovered ([#4980](https://github.com/can1357/oh-my-pi/issues/4980)). +- Fixed xAI SuperGrok multi-account rotation to correctly treat HTTP 403 credit exhaustion and spending limit errors as usage limits, triggering a credential rotation to a sibling account. +- Fixed error classification for AWS credential-resolution failures (AwsCredentialsError) to correctly map them as authentication failures. +- Fixed OpenAI-compatible chat-completions streams to preserve vLLM-style trailing cached-token usage chunks, ensuring accurate cacheRead and billable input session statistics. +- Fixed xai-oauth/grok-4.5 Responses requests to omit the unsupported reasoning.summary field while preserving the reasoning.effort payload. +- Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks once live usage indicates recovery. ## [16.3.15] - 2026-07-09 diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 02e7eee1e..0e17c4516 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,25 +2,21 @@ ## [Unreleased] +### Breaking Changes + +- Redesigned reasoning effort ladders to be wire-exact, removing the shifted five-tier effort mapping. Models now expose exactly the effort tiers their upstream APIs accept, mapped 1:1. Removed SHIFTED_FIVE_TIER_EFFORT_MAP, ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER, and per-host xhigh-to-max alias maps. Selecting an unsupported tier now automatically clamps down via clampThinkingLevelForModel. Devin effort routing is now mapped 1:1 onto per-tier siblings. + ### Added -- Added Grok 4.5 model family -- Added support for Dolphin Mistral 24b Venice Edition -- Added GLM5.2-Fast model -- Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) -- Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). -- Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. -- Added `Effort.Max` ("max") as a first-class user-facing thinking level above `xhigh`. +- Added support for new models: Grok 4.5 family, Dolphin Mistral 24b Venice Edition, GLM5.2-Fast, and Zenmux variants for GPT-5.6 (Luna, Sol, and Terra). +- Added Novita as a model provider, including public catalog discovery, pricing, limits, modality, reasoning, and tool metadata. +- Added useResponsesLite to Model and ModelSpec to support the Responses Lite transport, enabled by default for the GPT-5.6 family. +- Added Effort.Max ("max") as a first-class user-facing thinking level above xhigh. ### Changed -- Standardized reasoning effort levels to use a wire-exact `max` tier across all model providers -- Refactored Devin model routing to support 1:1 mapping for the `max` effort tier -- Normalized stale Ollama model configurations to the new wire-exact effort ladder - -- Updated costs and context windows for various models in the catalog -- **Breaking**: Effort ladders are now wire-exact and the shifted five-tier effort mapping is gone. Models expose exactly the effort tiers their wire accepts, mapped 1:1: GPT-5.6+ and Anthropic adaptive models with a real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5) expose `low..max`; legacy adaptive models (Opus 4.6 and all Bedrock adaptive) expose `low/medium/high/max`; Sonnet/Haiku 4.6 expose `low/medium/high`; GLM-5.2 on Z.ai/Zhipu/Umans/Ollama Cloud/Baseten and Sakana Fugu and DeepSeek expose `high/max`; local Ollama reasoning models expose `low/medium/high/max`; Fire Pass Kimi exposes `low..max` with distinct `xhigh` and `max` budgets. Removed `SHIFTED_FIVE_TIER_EFFORT_MAP`, `ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER`, and the per-host `xhigh -> "max"` alias maps; selecting a tier a model lacks clamps down via `clampThinkingLevelForModel`. Devin effort routing is now 1:1 onto per-tier siblings (`max -> -max`; families without a `-max` sibling top out at `xhigh`; the fake `minimal -> -low` fallback is gone). Regenerated `models.json`. -- Changed `fillThinkingWireDefaults` to re-derive the effort map whenever the model-defined ladder disagrees with cached metadata, so stale cached surfaces from the shifted-map era normalize to the new wire-exact shape on every `buildModel`. +- Standardized reasoning effort levels to use a wire-exact max tier across all model providers, including Devin routing and Ollama configurations. +- Updated costs and context windows for various models in the catalog. ## [16.3.15] - 2026-07-09 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 214127a96..34e214b90 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,36 +4,27 @@ ### Breaking Changes -- Renamed the bundled agent `explore` to `scout` (including renaming its configuration keys, prompt files, and task definitions). Any configurations, allowlists, or invocations referencing `explore` must now use `scout`. +- Renamed the bundled agent explore to scout, including its configuration keys, prompt files, and task definitions. Any configurations, allowlists, or invocations referencing explore must now use scout. ### Added -- Added `max` as a native, first-class thinking tier for supported models -- Added `thinkingBudgets.max` configuration setting -- Updated terminal theme to support optional `thinkingMax` border color and icons - -- Added a real `max` thinking level above `xhigh` with an optional `thinkingMax` theme border color (falls back to `thinkingXhigh`). `max` owns the top status-line icons (`◉` unicode, fire nerd-font, `[max]` ascii); the nerd-font preset uses an empty-to-full battery ramp for `minimal` through `xhigh` and shuffle while automatic effort is unresolved. `max` appears in cycling, selectors, `--thinking`, `:max` model suffixes, role scopes, settings, and completions on models that genuinely support it; on other models it clamps down like any unsupported tier. - -### Changed - -- Renamed the bundled agent `explore` to `scout` (including all internal references, prompt definitions, and task tool configurations). Any custom configurations or task invocations referencing `explore` must now use `scout`. -- Changed `max` from a parse-time alias of `xhigh` to a distinct level everywhere (CLI flag, `:max` suffix, `defaultThinkingLevel`, ACP/RPC): the effort a model receives is now exactly the tier its wire supports, with no shifted remapping. Ultrathink now requests `max` (clamped per model); automatic thinking still tops out at `xhigh`. Added `thinkingBudgets.max` (default 32768). -- Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) +- Added a native, first-class max thinking tier for supported models, including a new thinkingBudgets.max configuration setting, support in CLI flags (--thinking, :max model suffixes), and terminal theme customization (thinkingMax border color and icons). ### Fixed -- Fixed interactive TUI sessions dying with `Unhandled rejection: Cannot set cwd while another same-realm JS runtime is running` after the JS eval worker fell back to the in-process inline path (commonly when the worker could not load `pi_natives`). Concurrent inline eval/browser runtimes now stamp cwd (including the saved `__omp_session__` state) without stealing the exclusive realm, WorkerCore `init` reports failures via `init-failed` instead of throwing out of the microtask path, and constructing a runtime while another same-realm run is live fails explicitly instead of clobbering its globals. ([#4907](https://github.com/can1357/oh-my-pi/pull/4907) by [@cexll](https://github.com/cexll)) -- Fixed compaction aborting instead of trying an authenticated fallback model when Amazon Bedrock credential resolution fails before a request is sent. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) -- Fixed full-context forks cold-missing OpenAI prompt caches by persisting an inherited provider prompt-cache key separately from the new OMP session id, adding `--prompt-cache-key` for explicit cache affinity, and dropping automatic inheritance when startup changes the model, thinking level, system prompt, or tool schema. ([#5035](https://github.com/can1357/oh-my-pi/issues/5035)) -- Fixed Codex advisor requests using local `-advisor` session labels as provider session IDs; advisors now use stable UUIDv7 provider identities while keeping labeled transcript names. ([#5040](https://github.com/can1357/oh-my-pi/issues/5040)) -- Fixed macOS stdio MCP servers launching in a detached session, so `xcrun mcpbridge` can trigger the TCC Apple Events permission prompt and complete startup. ([#4987](https://github.com/can1357/oh-my-pi/issues/4987)) -- Fixed the ask tool timeout so it auto-selects the recommended option even when the UI selector does not settle on its own. ([#4995](https://github.com/can1357/oh-my-pi/issues/4995)) -- Fixed LSP workspace diagnostics for Go workspaces so roots with `go.work` are recognized and every `go.work use` module is included in the `go build` package patterns. ([#5038](https://github.com/can1357/oh-my-pi/issues/5038)) -- Fixed interactive OAuth login success messages waiting on model discovery; `/login xai-oauth` now reports saved credentials immediately while model metadata refreshes in the background. ([#4989](https://github.com/can1357/oh-my-pi/issues/4989)) -- Fixed Windows bash tool crashes when an explicit timeout fires while a piped command is still streaming output; the JavaScript fallback now reports the timeout without also aborting the native timeout signal. ([#5021](https://github.com/can1357/oh-my-pi/issues/5021)) -- Fixed subagent `yield` tool calls being discarded when the soft request budget hard-aborted the same assistant turn before the yield result event landed. ([#5006](https://github.com/can1357/oh-my-pi/issues/5006)) -- Fixed `--tools` filtering in interactive sessions disabling deferred MCP tools; MCP tools discovered from configured servers now stay active when the flag limits only built-in tools. ([#5013](https://github.com/can1357/oh-my-pi/issues/5013)) -- Fixed kept-alive task subagents entering a repeated provider-call loop after an IRC wake and terminal `yield`. ([#4963](https://github.com/can1357/oh-my-pi/issues/4963)) +- Fixed a memory leak (large retained JavaScriptCore heaps) in the TUI during session transcript rebuilds and refreshes by properly handling snapcompact archive image frames. +- Fixed a crash in interactive TUI sessions (Cannot set cwd while another same-realm JS runtime is running) when the JS evaluation worker falls back to the in-process inline path. +- Fixed compaction aborting when Amazon Bedrock credential resolution fails, ensuring it now falls back to trying an authenticated model. +- Improved OpenAI prompt cache hit rates for full-context forks by persisting inherited provider prompt-cache keys separately from session IDs, and added a --prompt-cache-key flag for explicit cache affinity. +- Fixed Codex advisor requests incorrectly using local session labels as provider session IDs, switching to stable UUIDv7 provider identities. +- Fixed macOS stdio MCP servers launching in detached sessions, allowing tools like xcrun mcpbridge to successfully trigger TCC Apple Events permission prompts. +- Fixed the ask tool timeout behavior to automatically select the recommended option if the UI selector does not settle. +- Fixed LSP workspace diagnostics for Go workspaces to correctly recognize go.work roots and include all specified modules in go build package patterns. +- Fixed interactive OAuth login (/login xai-oauth) delaying success messages; credentials are now reported immediately while model metadata refreshes in the background. +- Fixed a crash in the Windows bash tool when a timeout occurs while a piped command is streaming output. +- Fixed subagent yield tool calls being discarded when a soft request budget aborts the assistant turn before the yield event completes. +- Fixed --tools filtering in interactive sessions incorrectly disabling deferred MCP tools from configured servers. +- Fixed kept-alive task subagents entering infinite provider-call loops after an IRC wake and terminal yield. ## [16.3.15] - 2026-07-09 diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 8106893df..149e4c8a5 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed resume/session-replace and resize-settle full paints blanking the live viewport before replaying the transcript, preventing flicker on terminals without effective synchronized output ([#5028](https://github.com/can1357/oh-my-pi/issues/5028)). +- Fixed terminal flickering during session resume, replacement, or resizing on terminals that do not support synchronized output. ## [16.3.14] - 2026-07-09 From b0f22caf8308421ac6bb2462c1a4b729cb0c182d Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 11:51:54 +0000 Subject: [PATCH 068/205] fix(providers): restored copilot business vision Honor GitHub Copilot /models vision support on Business and Enterprise endpoints and remove the stale snapcompact non-personal-host block. Fixes #4779 --- packages/catalog/CHANGELOG.md | 4 ++ .../src/provider-models/openai-compat.ts | 22 +++---- packages/catalog/src/wire/github-copilot.ts | 8 +-- .../test/github-copilot-model-limits.test.ts | 58 ++++++++++++++----- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/session/snapcompact-inline.ts | 22 +------ .../test/snapcompact-inline.test.ts | 51 +++++++--------- 7 files changed, 84 insertions(+), 85 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 42e39218a..03e7fe0d6 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed GitHub Copilot Business and Enterprise discovery preserving vision-capable models from `/models` instead of downgrading them to text-only. ([#4779](https://github.com/can1357/oh-my-pi/issues/4779)) + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index d43a20143..b4a7ba007 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -16,12 +16,7 @@ import { getBundledModels } from "../models"; import type { Api, FetchImpl, Model, ModelSpec, OpenAICompat, Provider, ThinkingConfig } from "../types"; import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils"; import { coreWeaveProjectHeaders } from "../wire/coreweave"; -import { - COPILOT_API_HEADERS, - getGitHubCopilotBaseUrl, - isPersonalGitHubCopilotBaseUrl, - parseGitHubCopilotApiKey, -} from "../wire/github-copilot"; +import { COPILOT_API_HEADERS, getGitHubCopilotBaseUrl, parseGitHubCopilotApiKey } from "../wire/github-copilot"; import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references"; const MODELS_DEV_URL = "https://models.dev/api.json"; @@ -3695,16 +3690,13 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana ? entry.name : (reference?.name ?? defaults.name); const api = inferCopilotApi(defaults.id); - // `supports.vision` reports the model's intrinsic capability, but - // the business/enterprise endpoints respond `400 vision is not - // supported` on image inputs. Only honour the flag for the - // canonical personal-Copilot host. const supportsVision = extractCopilotSupportsVision(entry); - const input: ModelSpec["input"] = isPersonalGitHubCopilotBaseUrl(baseUrl) - ? supportsVision - ? ["text", "image"] - : (reference?.input ?? defaults.input) - : ["text"]; + const input: ModelSpec["input"] = + supportsVision === undefined + ? (reference?.input ?? defaults.input) + : supportsVision + ? ["text", "image"] + : ["text"]; // With COPILOT_API_HEADERS the served window is the long-context // ceiling; the default tier ends at token_prices.default.context_max // prompt tokens. Cap the base entry to the default tier — the long diff --git a/packages/catalog/src/wire/github-copilot.ts b/packages/catalog/src/wire/github-copilot.ts index 0cb935e7f..8f0a03a3b 100644 --- a/packages/catalog/src/wire/github-copilot.ts +++ b/packages/catalog/src/wire/github-copilot.ts @@ -45,13 +45,7 @@ export function isPublicGitHubHost(host: string): boolean { return PUBLIC_GITHUB_HOSTS.has(host.trim().toLowerCase()); } -/** - * Canonical personal-Copilot API host. The business - * (`api.business.githubcopilot.com`) and enterprise (`copilot-api.{domain}`) - * endpoints respond with HTTP 400 "vision is not supported" on image inputs, - * so catalog discovery and capability gates MUST honour the upstream's - * `supports.vision` flag only for this exact base URL. - */ +/** Canonical personal-Copilot API host. */ export const PERSONAL_GITHUB_COPILOT_BASE_URL = "https://api.githubcopilot.com" as const; /** `true` when the resolved base URL is the canonical personal-Copilot host. */ diff --git a/packages/catalog/test/github-copilot-model-limits.test.ts b/packages/catalog/test/github-copilot-model-limits.test.ts index d68b49c3d..b6ecfe3d4 100644 --- a/packages/catalog/test/github-copilot-model-limits.test.ts +++ b/packages/catalog/test/github-copilot-model-limits.test.ts @@ -526,10 +526,7 @@ describe("github copilot vision endpoint policy", () => { enterpriseUrl: "ghe.example.com", }); - it("strips vision when discovery resolves to the business endpoint, even though upstream reports it", async () => { - // `api.business.githubcopilot.com` responds `400 vision is not supported` - // on image inputs (issue #3387), so the catalog MUST ignore the upstream's - // `supports.vision = true` flag for non-personal hosts. + it("keeps vision when discovery resolves to the business endpoint and upstream reports it", async () => { const { models } = await discoverCopilotModels( { data: [ @@ -548,10 +545,10 @@ describe("github copilot vision endpoint policy", () => { ); const model = models.find(candidate => candidate.id === "claude-sonnet-4.6"); expect(model?.baseUrl).toBe("https://api.business.githubcopilot.com"); - expect(model?.input).toEqual(["text"]); + expect(model?.input).toEqual(["text", "image"]); }); - it("strips vision when discovery resolves to an enterprise host", async () => { + it("keeps vision when discovery resolves to an enterprise host and upstream reports it", async () => { const { models } = await discoverCopilotModels( { data: [ @@ -570,7 +567,42 @@ describe("github copilot vision endpoint policy", () => { ); const model = models.find(candidate => candidate.id === "claude-sonnet-4.6"); expect(model?.baseUrl).toBe("https://copilot-api.ghe.example.com"); - expect(model?.input).toEqual(["text"]); + expect(model?.input).toEqual(["text", "image"]); + }); + + it("maps explicit upstream vision false to text-only on non-personal Copilot endpoints", async () => { + for (const endpoint of [ + { + apiKey: businessApiKey, + baseUrl: "https://api.business.githubcopilot.com", + token: "ghu_business_token", + }, + { + apiKey: enterpriseApiKey, + baseUrl: "https://copilot-api.ghe.example.com", + token: "ghu_enterprise_token", + }, + ]) { + const { models } = await discoverCopilotModels( + { + data: [ + tieredCopilotEntry({ + id: "claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + window: 200_000, + maxOutput: 32_000, + vision: false, + }), + ], + }, + endpoint.apiKey, + endpoint.baseUrl, + endpoint.token, + ); + const model = models.find(candidate => candidate.id === "claude-sonnet-4.6"); + expect(model?.baseUrl).toBe(endpoint.baseUrl); + expect(model?.input).toEqual(["text"]); + } }); it("keeps vision on the canonical personal Copilot endpoint", async () => { @@ -590,11 +622,11 @@ describe("github copilot vision endpoint policy", () => { expect(model?.input).toEqual(["text", "image"]); }); - it("downgrades the merged Model to text-only when business discovery overrides a vision-capable bundled reference", async () => { - // Bundled `claude-sonnet-4.6` ships with `input=['text','image']` and the - // canonical baseUrl. Discovery against the business host hands back a - // dynamic entry with the business baseUrl; the merge MUST honour the - // dynamic side's text-only capability instead of OR-upgrading. + it("keeps the merged Model image-capable when business discovery confirms a vision-capable bundled reference", async () => { + // Bundled `claude-sonnet-4.6` ships with `input=['text','image']`. + // Discovery against the business host confirms the same upstream vision + // capability; the full manager merge must preserve image input instead + // of downgrading solely because the baseUrl is non-personal. const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-copilot-vision-")); try { const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { @@ -629,7 +661,7 @@ describe("github copilot vision endpoint policy", () => { const { models } = await manager.refresh("online"); const model = models.find(candidate => candidate.id === "claude-sonnet-4.6"); expect(model?.baseUrl).toBe("https://api.business.githubcopilot.com"); - expect(model?.input).toEqual(["text"]); + expect(model?.input).toEqual(["text", "image"]); } finally { await fs.rm(tempDir, { recursive: true, force: true }); } diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..dcf8904cb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed snapcompact inline imaging for GitHub Copilot Business and Enterprise models that advertise image input, removing the stale non-personal-host block. ([#4779](https://github.com/can1357/oh-my-pi/issues/4779)) + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/session/snapcompact-inline.ts b/packages/coding-agent/src/session/snapcompact-inline.ts index a6d6f988f..7345053c1 100644 --- a/packages/coding-agent/src/session/snapcompact-inline.ts +++ b/packages/coding-agent/src/session/snapcompact-inline.ts @@ -16,7 +16,6 @@ import { countTokens } from "@oh-my-pi/pi-agent-core"; import type { Context, ImageContent, Model, TextContent, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai"; -import { isPersonalGitHubCopilotBaseUrl } from "@oh-my-pi/pi-catalog/wire/github-copilot"; import * as snapcompact from "@oh-my-pi/snapcompact"; import contextFramesNote from "../prompts/system/snapcompact-context-frames-note.md" with { type: "text" }; import contextStub from "../prompts/system/snapcompact-context-stub.md" with { type: "text" }; @@ -78,19 +77,6 @@ function passesSavingsGate(frames: number, shape: snapcompact.Shape, textTokens: return frames * shape.frameTokenEstimate <= textTokens * SAVINGS_MARGIN; } -/** - * The model is vision-capable for the endpoint we're actually about to hit. - * GitHub Copilot business and enterprise hosts respond `400 vision is not - * supported` on image inputs (issue #3387), so even if a stale cached spec - * still advertises `["text","image"]` we MUST not rasterize transcripts when - * the resolved `baseUrl` is non-personal. - */ -function canSendImages(model: Model): boolean { - if (!model.input.includes("image")) return false; - if (model.provider === "github-copilot" && !isPersonalGitHubCopilotBaseUrl(model.baseUrl)) return false; - return true; -} - interface SystemPromptImageTarget { scope: Exclude; text: string; @@ -291,7 +277,7 @@ export function estimateInlineSavings(input: { messages: readonly InlineMessageView[]; }): SnapcompactSavingsEstimate { const { options, model } = input; - if (!model || !canSendImages(model)) { + if (!model?.input.includes("image")) { return { visionCapable: false, savedTokens: 0 }; } @@ -430,10 +416,8 @@ export class SnapcompactInlineTransformer { async transform(context: Context, model: Model): Promise { // Vision gate: providers silently DROP images on text-only models — - // rendering would lose the content entirely. Also short-circuits when the - // resolved endpoint rejects vision regardless of the model's input list - // (issue #3387: Copilot business endpoint). - if (!canSendImages(model)) return context; + // rendering would lose the content entirely. + if (!model.input.includes("image")) return context; const shape = snapcompact.resolveShape(model, this.options.shape); const budget = snapcompact.providerImageBudget(model.provider) - countContextImages(context); diff --git a/packages/coding-agent/test/snapcompact-inline.test.ts b/packages/coding-agent/test/snapcompact-inline.test.ts index 36321c91e..0a30d61d4 100644 --- a/packages/coding-agent/test/snapcompact-inline.test.ts +++ b/packages/coding-agent/test/snapcompact-inline.test.ts @@ -101,41 +101,30 @@ describe("SnapcompactInlineTransformer", () => { expect(await transformer.transform(context, makeModel({ input: ["text"] }))).toBe(context); }); - it("is a no-op for Copilot business/enterprise endpoints even when the model claims vision (#3387)", async () => { + it("treats Copilot business/enterprise endpoints as vision-capable when the model input includes image", async () => { const transformer = new SnapcompactInlineTransformer( withTestShape({ renderSystemPrompt: "all", renderToolResults: true }), ); - const context = makeContext(); - const business = makeModel({ - provider: "github-copilot", - baseUrl: "https://api.business.githubcopilot.com", - input: ["text", "image"], - }); - expect(await transformer.transform(context, business)).toBe(context); - expect( - estimateInlineSavings({ - options: withTestShape({ renderSystemPrompt: "all", renderToolResults: true }), - model: business, - systemPrompt: context.systemPrompt ?? [], - messages: context.messages, - }), - ).toEqual({ visionCapable: false, savedTokens: 0 }); + for (const baseUrl of ["https://api.business.githubcopilot.com", "https://copilot-api.ghe.example.com"]) { + const context = makeContext(); + const model = makeModel({ + provider: "github-copilot", + baseUrl, + input: ["text", "image"], + }); + expect( + estimateInlineSavings({ + options: withTestShape({ renderSystemPrompt: "all", renderToolResults: true }), + model, + systemPrompt: context.systemPrompt ?? [], + messages: context.messages, + }).visionCapable, + ).toBe(true); - const enterprise = makeModel({ - provider: "github-copilot", - baseUrl: "https://copilot-api.ghe.example.com", - input: ["text", "image"], - }); - expect(await transformer.transform(context, enterprise)).toBe(context); - - const personal = makeModel({ - provider: "github-copilot", - baseUrl: "https://api.githubcopilot.com", - input: ["text", "image"], - }); - const result = await transformer.transform(context, personal); - expect(result).not.toBe(context); - expect(imageCount(result)).toBeGreaterThan(0); + const result = await transformer.transform(context, model); + expect(result).not.toBe(context); + expect(imageCount(result)).toBeGreaterThan(0); + } }); it("images large historical tool results, keeping small and most-recent ones as text", async () => { From 520f6f9bcc56ed6b1325f37bcc539e2faabb791f Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 13:54:11 +0200 Subject: [PATCH 069/205] feat(catalog): standardized reasoning effort configuration - Transitioned DeepSeek models to an explicit `[High, Max]` effort ladder, removing stale alias maps. - Updated the OpenAI compatibility layer to enforce authoritative `supportsReasoningEffort` and `omitReasoningEffort` flags. - Synchronized `models.json` definitions to reflect accurate reasoning effort capabilities across the catalog. --- packages/ai/test/issue-1207-repro.test.ts | 26 ++++++++----------- .../test/openai-responses-openrouter.test.ts | 4 +-- packages/catalog/CHANGELOG.md | 1 + packages/catalog/src/models.json | 15 +++++++---- .../src/provider-models/openai-compat.ts | 9 ++++--- .../coding-agent/src/modes/theme/theme.ts | 12 ++++----- 6 files changed, 36 insertions(+), 31 deletions(-) diff --git a/packages/ai/test/issue-1207-repro.test.ts b/packages/ai/test/issue-1207-repro.test.ts index f06a566b6..bf6a572af 100644 --- a/packages/ai/test/issue-1207-repro.test.ts +++ b/packages/ai/test/issue-1207-repro.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { Context, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { type } from "arktype"; @@ -27,7 +28,7 @@ function abortedSignal(): AbortSignal { async function capturePayload( model: Model<"openai-completions">, tools?: Tool[], - reasoning: "minimal" | "xhigh" = "minimal", + reasoning: "high" | "max" = "high", ): Promise> { const { promise, resolve } = Promise.withResolvers(); streamOpenAICompletions(model, contextWithTools(tools), { @@ -65,25 +66,20 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { expect(compat.supportsToolChoice).toBe(false); expect(compat.maxTokensField).toBe("max_tokens"); expect(compat.extraBody).toEqual({ thinking: { type: "enabled" } }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + // DeepSeek's reasoning_effort is the honest wire-exact high/max pair; + // no synthetic lower tiers, no alias map. + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); - it("merges partial user reasoning maps with DeepSeek defaults in thinking metadata", () => { + it("drops user reasoning map entries outside the honest DeepSeek ladder", () => { const model = customDeepseekFlash(); expect(model.compat.supportsToolChoice).toBe(false); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - xhigh: "max", - }); + // The stale user `xhigh` alias targets a tier the wire-exact + // [high, max] ladder no longer exposes, so it is filtered out. + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); it("omits tool_choice but preserves documented reasoning when tools are present", async () => { diff --git a/packages/ai/test/openai-responses-openrouter.test.ts b/packages/ai/test/openai-responses-openrouter.test.ts index 019d2d538..71a9c7396 100644 --- a/packages/ai/test/openai-responses-openrouter.test.ts +++ b/packages/ai/test/openai-responses-openrouter.test.ts @@ -211,7 +211,7 @@ describe("OpenRouter pseudo API dual-surface request parity", () => { stream: true, stream_options: { include_usage: true }, store: false, - reasoning: { effort: "xhigh" }, + reasoning: { effort: "high" }, provider: routing, }); expect(responsesBody).toEqual({ @@ -220,7 +220,7 @@ describe("OpenRouter pseudo API dual-surface request parity", () => { stream: true, input: [{ role: "user", content: [{ type: "input_text", text: "ping" }] }], store: false, - reasoning: { effort: "xhigh", summary: "auto" }, + reasoning: { effort: "high", summary: "auto" }, prompt_cache_key: "workflow-123", session_id: "workflow-123", provider: routing, diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 0e17c4516..3c91836b9 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -15,6 +15,7 @@ ### Changed +- Enabled reasoning effort controls for Grok 4.5 and updated support flags for additional Grok variants - Standardized reasoning effort levels to use a wire-exact max tier across all model providers, including Devin routing and Ollama configurations. - Updated costs and context windows for various models in the catalog. diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 474e62114..7eecd2bd0 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -87161,7 +87161,8 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": true + "omitReasoningEffort": true, + "supportsReasoningEffort": false } }, "grok-4.20-0309-reasoning": { @@ -87230,7 +87231,8 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": false + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-4.3": { @@ -87271,7 +87273,8 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": false + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-4.5": { @@ -87312,7 +87315,8 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": true + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-build": { @@ -87397,7 +87401,8 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": true + "omitReasoningEffort": true, + "supportsReasoningEffort": false } } }, diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 421a6a404..8952fcb78 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1199,18 +1199,21 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const; // The `minimal -> low` effort clamp (XAI_REASONING_EFFORT_MAP) is always // merged in so dynamic-fetched models — which arrive without curated // compat keys — still get the clamp applyResponsesReasoningParams expects. +// The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is +// authoritative: a stale flag on `base` (previous snapshot or dynamic fetch) +// must not outlive an allowlist change in identity/family.ts. function mergeCuratedIntoModel( base: ModelSpec<"openai-responses">, curated: XAICuratedModel, ): ModelSpec<"openai-responses"> { - const effort = curated.supportsReasoningEffort; + const effortCapable = curated.supportsReasoningEffort ?? isGrokReasoningEffortCapable(curated.id); const compat = { ...(base.compat ?? {}), reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) }, includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false, filterReasoningHistory: base.compat?.filterReasoningHistory ?? true, - omitReasoningEffort: base.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(base.id), - ...(effort === undefined ? {} : { supportsReasoningEffort: effort }), + omitReasoningEffort: !effortCapable, + supportsReasoningEffort: effortCapable, }; return { ...base, diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index ca0a6397f..fc551e692 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -641,12 +641,12 @@ const NERD_SYMBOLS: SymbolMap = { "icon.mic": "\uf130", // Compaction divider - fa-camera-retro "icon.camera": "\uf083", - // Thinking levels — empty-to-full battery ramp, then fire. - "thinking.minimal": "\u{F244} min", - "thinking.low": "\u{F243} low", - "thinking.medium": "\u{F242} med", - "thinking.high": "\u{F241} high", - "thinking.xhigh": "\u{F240} xhi", + // Thinking levels — increasing circle slices, with fire reserved for max. + "thinking.minimal": "\u{F0A9E} min", + "thinking.low": "\u{F0A9F} low", + "thinking.medium": "\u{F0AA1} med", + "thinking.high": "\u{F0AA3} high", + "thinking.xhigh": "\u{F0AA5} xhi", "thinking.max": "\u{F06D} max", // Auto mode uses shuffle until the model resolves its thinking level. "thinking.autoPending": "\u{F074}", From e8add6101648cd15b2b7ab1dc9e5c29115f9f99c Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 14:04:32 +0200 Subject: [PATCH 070/205] fix(debug): stopped native fallback for missing delve - Kept missing language-specific adapters from falling through to native debuggers. - Resolved nested launch roots before session-local binaries and PATH, including explicit adapters and go.work workspaces. - Added actionable install/configuration errors and deterministic regression coverage. Fixes #5037 --- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/src/dap/config.ts | 181 ++++++++++++++---- packages/coding-agent/src/dap/defaults.json | 2 +- packages/coding-agent/src/lsp/config.ts | 40 ++-- .../coding-agent/src/prompts/tools/debug.md | 8 +- packages/coding-agent/src/tools/debug.ts | 52 +++-- .../test/debug/dap-config.test.ts | 154 ++++++++++++++- .../test/debug/dap-launch-failures.test.ts | 108 ++++++++++- 8 files changed, 465 insertions(+), 82 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 34e214b90..15ad4eaa6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,6 +5,7 @@ ### Breaking Changes - Renamed the bundled agent explore to scout, including its configuration keys, prompt files, and task definitions. Any configurations, allowlists, or invocations referencing explore must now use scout. +- Changed the public `selectLaunchAdapter()` result from `DapResolvedAdapter | null` to `LaunchAdapterSelection`; callers must handle `adapter`, `unavailable`, and `none` outcomes. ### Added @@ -12,6 +13,7 @@ ### Fixed +- Fixed Go debug launches falling back to native debuggers when Delve is unavailable; nested modules and `go.work` workspaces now resolve local Delve adapters before PATH, newly installed adapters are detected without restart, and missing adapter errors include install or configuration guidance. ([#5037](https://github.com/can1357/oh-my-pi/issues/5037)) - Fixed a memory leak (large retained JavaScriptCore heaps) in the TUI during session transcript rebuilds and refreshes by properly handling snapcompact archive image frames. - Fixed a crash in interactive TUI sessions (Cannot set cwd while another same-realm JS runtime is running) when the JS evaluation worker falls back to the in-process inline path. - Fixed compaction aborting when Amazon Bedrock credential resolution fails, ensuring it now falls back to trying an authenticated model. diff --git a/packages/coding-agent/src/dap/config.ts b/packages/coding-agent/src/dap/config.ts index bb5748fd7..0aaa27eea 100644 --- a/packages/coding-agent/src/dap/config.ts +++ b/packages/coding-agent/src/dap/config.ts @@ -1,7 +1,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { isRecord, logger } from "@oh-my-pi/pi-utils"; +import { isRecord, logger, WhichCachePolicy } from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; import { getConfigDirPaths } from "../config"; import { getPreloadedPluginRoots } from "../discovery/helpers"; @@ -9,7 +9,7 @@ import { hasRootMarkers, resolveCommand } from "../lsp/config"; import DEFAULTS from "./defaults.json" with { type: "json" }; import type { DapAdapterConfig, DapResolvedAdapter } from "./types"; -const EXTENSIONLESS_DEBUGGER_ORDER = ["gdb", "lldb-dap"] as const; +const EXTENSIONLESS_DEBUGGER_ORDER: readonly string[] = ["gdb", "lldb-dap"]; interface NormalizedConfig { adapters: Record; @@ -188,10 +188,18 @@ function resolveAdapterFromConfig( adapterName: string, configs: Record, cwd: string, + localRoots?: readonly string[], ): DapResolvedAdapter | null { const config = configs[adapterName]; if (!config) return null; - const resolvedCommand = resolveCommand(normalizeCommandForCwd(config.command, cwd), cwd); + const normalizedCommand = normalizeCommandForCwd(config.command, cwd); + const commandIsBare = + !path.isAbsolute(config.command) && !config.command.includes("/") && !config.command.includes("\\"); + const resolvedCommand = resolveCommand(normalizedCommand, cwd, { + cache: WhichCachePolicy.Fresh, + PATH: process.env.PATH, + localRoots: commandIsBare ? localRoots : undefined, + }); if (!resolvedCommand) return null; return { name: adapterName, @@ -219,33 +227,125 @@ export function getAvailableAdapters(cwd: string): DapResolvedAdapter[] { .filter((adapter): adapter is DapResolvedAdapter => adapter !== null); } -function getMatchingAdapters(program: string, cwd: string): DapResolvedAdapter[] { - const extension = path.extname(program).toLowerCase(); - const available = getAvailableAdapters(cwd); - if (!extension) { - // For extensionless binaries, only consider native debuggers (gdb, lldb-dap) - // or adapters that match by root markers. Don't silently fall back to - // unrelated adapters like debugpy for a C binary. - const nativeDebuggers: ReadonlySet = new Set(EXTENSIONLESS_DEBUGGER_ORDER); - return available.filter( - adapter => - nativeDebuggers.has(adapter.name) || - (adapter.rootMarkers.length > 0 && hasRootMarkers(cwd, adapter.rootMarkers)), - ); - } - const exactMatches = available.filter(adapter => adapter.fileTypes.includes(extension)); - if (exactMatches.length > 0) { - return exactMatches; - } - return available; +/** Launch adapter selection, including a configured adapter whose command is unavailable. */ +export type LaunchAdapterSelection = + | { kind: "adapter"; adapter: DapResolvedAdapter } + | { kind: "unavailable"; adapterName: string; command: string } + | { kind: "none" }; + +interface LaunchAdapterCandidate { + name: string; + rootDir: string | null; } -function sortAdaptersForLaunch(program: string, cwd: string, adapters: DapResolvedAdapter[]): DapResolvedAdapter[] { +function findRootMarkerInLaunchAncestry( + program: string, + cwd: string, + markers: string[], + programKind: LaunchProgramKind, +): string | null { + if (markers.length === 0) return null; + let dir = programKind === "directory" ? path.resolve(cwd, program) : path.dirname(path.resolve(cwd, program)); + while (true) { + if (hasRootMarkers(dir, markers)) return dir; + const parent = path.dirname(dir); + if (parent === dir) return null; + dir = parent; + } +} + +function resolveAdapterForLaunch( + adapterName: string, + configs: Record, + cwd: string, + rootDir: string | null, +): DapResolvedAdapter | null { + const localRoots = rootDir && rootDir !== cwd ? [rootDir, cwd] : undefined; + return resolveAdapterFromConfig(adapterName, configs, cwd, localRoots); +} + +function unavailableAdapter( + candidate: LaunchAdapterCandidate, + configs: Record, +): LaunchAdapterSelection { + const config = configs[candidate.name]; + if (!config) return { kind: "none" }; + return { kind: "unavailable", adapterName: candidate.name, command: config.command }; +} + +function selectAutomaticLaunchAdapter( + program: string, + cwd: string, + programKind: LaunchProgramKind, + configs: Record, +): LaunchAdapterSelection { + const extension = path.extname(program).toLowerCase(); + if (extension) { + const configured: LaunchAdapterCandidate[] = []; + const available: DapResolvedAdapter[] = []; + for (const name in configs) { + const config = configs[name]; + if (!config || !(config.fileTypes ?? []).includes(extension)) continue; + const rootDir = findRootMarkerInLaunchAncestry(program, cwd, config.rootMarkers ?? [], programKind); + configured.push({ name, rootDir }); + const adapter = resolveAdapterForLaunch(name, configs, cwd, rootDir); + if (adapter) available.push(adapter); + } + const selected = sortAdaptersForLaunch(program, cwd, programKind, available)[0]; + if (selected) return { kind: "adapter", adapter: selected }; + const rootMatch = configured.find(candidate => candidate.rootDir !== null); + const unavailable = rootMatch ?? configured[0]; + if (unavailable) return unavailableAdapter(unavailable, configs); + } + + const available: DapResolvedAdapter[] = []; + const rootMatches: LaunchAdapterCandidate[] = []; + const directoryMatches: LaunchAdapterCandidate[] = []; + for (const name in configs) { + const config = configs[name]; + if (!config) continue; + const rootDir = findRootMarkerInLaunchAncestry(program, cwd, config.rootMarkers ?? [], programKind); + const candidate = { name, rootDir }; + if (rootDir) { + rootMatches.push(candidate); + if (config.acceptsDirectoryProgram === true) directoryMatches.push(candidate); + } + if (!EXTENSIONLESS_DEBUGGER_ORDER.includes(name) && !rootDir) continue; + const adapter = resolveAdapterForLaunch(name, configs, cwd, rootDir); + if (adapter) available.push(adapter); + } + + if (programKind === "directory" && directoryMatches.length > 0) { + const matchingNames = new Set(directoryMatches.map(candidate => candidate.name)); + const directoryAdapters = available.filter( + adapter => adapter.acceptsDirectoryProgram && matchingNames.has(adapter.name), + ); + const selected = sortAdaptersForLaunch(program, cwd, programKind, directoryAdapters)[0]; + if (selected) return { kind: "adapter", adapter: selected }; + const unavailable = directoryMatches[0]; + return unavailable ? unavailableAdapter(unavailable, configs) : { kind: "none" }; + } + + const directoryAdapters = + programKind === "directory" ? available.filter(adapter => adapter.acceptsDirectoryProgram) : available; + const candidates = directoryAdapters.length > 0 ? directoryAdapters : available; + const selected = sortAdaptersForLaunch(program, cwd, programKind, candidates)[0]; + if (selected) return { kind: "adapter", adapter: selected }; + const unavailable = rootMatches[0]; + return unavailable ? unavailableAdapter(unavailable, configs) : { kind: "none" }; +} + +function sortAdaptersForLaunch( + program: string, + cwd: string, + programKind: LaunchProgramKind, + adapters: DapResolvedAdapter[], +): DapResolvedAdapter[] { const extension = path.extname(program).toLowerCase(); const rootAware = adapters.map(adapter => ({ adapter, hasExtensionMatch: extension.length > 0 && adapter.fileTypes.includes(extension), - hasRootMatch: adapter.rootMarkers.length > 0 && hasRootMarkers(cwd, adapter.rootMarkers), + hasRootMatch: findRootMarkerInLaunchAncestry(program, cwd, adapter.rootMarkers, programKind) !== null, })); rootAware.sort((left, right) => { if (left.hasExtensionMatch !== right.hasExtensionMatch) { @@ -254,36 +354,33 @@ function sortAdaptersForLaunch(program: string, cwd: string, adapters: DapResolv if (left.hasRootMatch !== right.hasRootMatch) { return left.hasRootMatch ? -1 : 1; } - const leftDebuggerRank = EXTENSIONLESS_DEBUGGER_ORDER.indexOf( - left.adapter.name as (typeof EXTENSIONLESS_DEBUGGER_ORDER)[number], - ); - const rightDebuggerRank = EXTENSIONLESS_DEBUGGER_ORDER.indexOf( - right.adapter.name as (typeof EXTENSIONLESS_DEBUGGER_ORDER)[number], - ); - const normalizedLeftRank = leftDebuggerRank === -1 ? Number.MAX_SAFE_INTEGER : leftDebuggerRank; - const normalizedRightRank = rightDebuggerRank === -1 ? Number.MAX_SAFE_INTEGER : rightDebuggerRank; - if (normalizedLeftRank !== normalizedRightRank) { - return normalizedLeftRank - normalizedRightRank; - } + const leftRank = EXTENSIONLESS_DEBUGGER_ORDER.indexOf(left.adapter.name); + const rightRank = EXTENSIONLESS_DEBUGGER_ORDER.indexOf(right.adapter.name); + const normalizedLeftRank = leftRank === -1 ? Number.MAX_SAFE_INTEGER : leftRank; + const normalizedRightRank = rightRank === -1 ? Number.MAX_SAFE_INTEGER : rightRank; + const rankDelta = normalizedLeftRank - normalizedRightRank; + if (rankDelta !== 0) return rankDelta; return left.adapter.name.localeCompare(right.adapter.name); }); return rootAware.map(entry => entry.adapter); } +/** Selects a launch adapter or reports why matching configuration cannot run. */ export function selectLaunchAdapter( program: string, cwd: string, adapterName?: string, programKind: LaunchProgramKind = "file", -): DapResolvedAdapter | null { +): LaunchAdapterSelection { + const configs = getAdapterConfigs(cwd); if (adapterName) { - return resolveAdapter(adapterName, cwd); + const config = configs[adapterName]; + if (!config) return { kind: "none" }; + const rootDir = findRootMarkerInLaunchAncestry(program, cwd, config.rootMarkers ?? [], programKind); + const adapter = resolveAdapterForLaunch(adapterName, configs, cwd, rootDir); + return adapter ? { kind: "adapter", adapter } : { kind: "unavailable", adapterName, command: config.command }; } - const matches = getMatchingAdapters(program, cwd); - const candidates = - programKind === "directory" ? matches.filter(adapter => adapter.acceptsDirectoryProgram) : matches; - const sorted = sortAdaptersForLaunch(program, cwd, candidates.length > 0 ? candidates : matches); - return sorted[0] ?? null; + return selectAutomaticLaunchAdapter(program, cwd, programKind, configs); } export function selectAttachAdapter(cwd: string, adapterName?: string, port?: number): DapResolvedAdapter | null { diff --git a/packages/coding-agent/src/dap/defaults.json b/packages/coding-agent/src/dap/defaults.json index 5ea74ef25..c33eb20ba 100644 --- a/packages/coding-agent/src/dap/defaults.json +++ b/packages/coding-agent/src/dap/defaults.json @@ -64,7 +64,7 @@ "connectMode": "socket", "languages": ["go"], "fileTypes": [".go"], - "rootMarkers": ["go.mod", "go.sum"], + "rootMarkers": ["go.mod", "go.sum", "go.work"], "acceptsDirectoryProgram": true, "launchDefaults": { "request": "launch", diff --git a/packages/coding-agent/src/lsp/config.ts b/packages/coding-agent/src/lsp/config.ts index e9b650bd6..336c113ba 100644 --- a/packages/coding-agent/src/lsp/config.ts +++ b/packages/coding-agent/src/lsp/config.ts @@ -1,7 +1,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { $which, isRecord, logger, pathIsWithin } from "@oh-my-pi/pi-utils"; +import { $which, isRecord, logger, pathIsWithin, type WhichOptions } from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; import { getConfigDirPaths } from "../config"; import { type ClaudePluginRoot, getPreloadedPluginRoots } from "../discovery/helpers"; @@ -249,7 +249,7 @@ const LOCAL_BIN_PATHS: Array<{ markers: string[]; binDir: string }> = [ { markers: ["Gemfile", "Gemfile.lock"], binDir: "vendor/bundle/bin" }, { markers: ["Gemfile", "Gemfile.lock"], binDir: "bin" }, // Go - check project-local bin - { markers: ["go.mod", "go.sum"], binDir: "bin" }, + { markers: ["go.mod", "go.sum", "go.work"], binDir: "bin" }, ]; const WINDOWS_LOCAL_EXECUTABLE_EXTENSIONS = [".exe", ".cmd", ".bat"] as const; @@ -267,6 +267,21 @@ function resolveLocalCommand(basePath: string): string | null { return null; } +function resolveCommandFromLocalRoot(command: string, cwd: string): string | null { + for (const { markers, binDir } of LOCAL_BIN_PATHS) { + if (!hasRootMarkers(cwd, markers)) continue; + const resolved = resolveLocalCommand(path.join(cwd, binDir, command)); + if (resolved) return resolved; + } + return null; +} + +/** Controls project-local and PATH executable lookup. */ +export interface ResolveCommandOptions extends Pick { + /** Ordered project roots checked before PATH; defaults to the command cwd. */ + localRoots?: readonly string[]; +} + /** * Resolve a command to an executable path. * Checks project-local bin directories first, then falls back to $PATH. @@ -275,20 +290,19 @@ function resolveLocalCommand(basePath: string): string | null { * @param cwd - Working directory to search from * @returns Absolute path to the executable, or null if not found */ -export function resolveCommand(command: string, cwd: string): string | null { - // Check local bin directories based on project markers - for (const { markers, binDir } of LOCAL_BIN_PATHS) { - if (hasRootMarkers(cwd, markers)) { - const localPath = path.join(cwd, binDir, command); - const resolvedLocalPath = resolveLocalCommand(localPath); - if (resolvedLocalPath) { - return resolvedLocalPath; - } +export function resolveCommand(command: string, cwd: string, options?: ResolveCommandOptions): string | null { + if (options?.localRoots) { + for (const root of options.localRoots) { + const resolved = resolveCommandFromLocalRoot(command, root); + if (resolved) return resolved; } + } else { + const resolved = resolveCommandFromLocalRoot(command, cwd); + if (resolved) return resolved; } - // Fall back to $PATH - return $which(command); + if (!options) return $which(command); + return $which(command, { cache: options.cache, PATH: options.PATH }); } interface ConfigSource { diff --git a/packages/coding-agent/src/prompts/tools/debug.md b/packages/coding-agent/src/prompts/tools/debug.md index eddbabaa8..3412066ae 100644 --- a/packages/coding-agent/src/prompts/tools/debug.md +++ b/packages/coding-agent/src/prompts/tools/debug.md @@ -2,7 +2,7 @@ Debugger access. - You SHOULD prefer this over bash for program state, breakpoints, stepping, thread inspection, or interrupting a running process. -- `action: "launch"` starts a session; `program` required, `adapter` optional. Python: `adapter: "debugpy"`, `program` = target `.py`, interpreter/script flags in `args`. +- `action: "launch"` starts a session; `program` required, `adapter` optional. Python: `program` = target `.py`, interpreter/script flags in `args`. Go: `program` = package directory, `.go` file, or compiled binary. - `action: "attach"` connects to a running process: `pid` (local), `port` (remote), `adapter` forces a specific debugger. - **Breakpoints**: `set_breakpoint`/`remove_breakpoint` with source (`file`+`line`) or function (`function`); optional `condition`. - **Flow control**: `continue` (resume), `step_over`/`step_in`/`step_out` (single-step), `pause` (interrupt a running program). @@ -11,7 +11,7 @@ Debugger access. - Only one active debug session at a time. -- Valid `adapter` values: `gdb`, `lldb-dap`, `python -m debugpy.adapter`, `dlv dap` (must be installed locally). -- `program` must be an executable file or debug target, not a directory or bare interpreter name. -- Python debugging requires `debugpy`; `pip install debugpy` if unavailable. +- `adapter` is a configured id: `gdb`, `lldb-dap`, `debugpy`, `dlv`, `rdbg`, or any `dap.json` entry; its command must be installed. +- `program` is a target path, not a shell command. Directories require a directory-capable adapter such as `dlv`. +- Python requires `debugpy` (`pip install debugpy`); Go requires Delve (`go install github.com/go-delve/delve/cmd/dlv@latest`); Ruby requires `rdbg` (`gem install debug`). diff --git a/packages/coding-agent/src/tools/debug.ts b/packages/coding-agent/src/tools/debug.ts index 3d0680d37..690243120 100644 --- a/packages/coding-agent/src/tools/debug.ts +++ b/packages/coding-agent/src/tools/debug.ts @@ -31,6 +31,7 @@ import { type DapThread, type DapVariable, dapSessionManager, + getAdapterConfigs, getAvailableAdapters, type LaunchProgramKind, resolveLaunchOverrides, @@ -50,6 +51,7 @@ import { formatStatusIcon, PREVIEW_LIMITS, replaceTabs, + shortenPath, TRUNCATE_LENGTHS, truncateToWidth, } from "./render-utils"; @@ -106,9 +108,9 @@ const debugActionSchema = type.enumerated( ); const debugSchema = type({ action: debugActionSchema, - "program?": type("string").describe("program path"), + "program?": type("string").describe("debug target path; Delve accepts Go package directories"), "args?": type("string[]").describe("program arguments"), - "adapter?": type("string").describe("debugger adapter (gdb, lldb-dap, debugpy, dlv)"), + "adapter?": type("string").describe("configured adapter id (gdb, lldb-dap, debugpy, dlv, rdbg, or dap.json entry)"), cwd: "string?", "file?": type("string").describe("source file"), "line?": type("number").describe("source line"), @@ -494,7 +496,33 @@ function buildOutcomeText(outcome: DapContinueOutcome, timeoutSec: number, verb: function getConfiguredAdapters(cwd: string): string { const adapters = getAvailableAdapters(cwd).map(adapter => adapter.name); - return adapters.length > 0 ? adapters.join(", ") : "none"; + const names = adapters.length > 0 ? adapters.join(", ") : "none"; + return truncateToWidth(replaceTabs(names), TRUNCATE_LENGTHS.LONG); +} + +const ADAPTER_UNAVAILABLE_MESSAGES: Readonly> = { + debugpy: "adapter 'debugpy' is not available: python not found in PATH", + dlv: "adapter 'dlv' is not available: install with 'go install github.com/go-delve/delve/cmd/dlv@latest'", + rdbg: "adapter 'rdbg' is not available: install with 'gem install debug'", +}; + +const ADAPTER_CANONICAL_COMMANDS: Readonly> = { + debugpy: "python", + dlv: "dlv", + rdbg: "rdbg", +}; + +function formatAdapterUnavailable(adapterName: string, command: string, cwd: string): string { + const displayName = truncateToWidth(replaceTabs(adapterName), TRUNCATE_LENGTHS.SHORT); + const canonicalCommand = ADAPTER_CANONICAL_COMMANDS[adapterName] ?? adapterName; + if (command !== canonicalCommand) { + const displayCommand = truncateToWidth(replaceTabs(shortenPath(command)), TRUNCATE_LENGTHS.CONTENT); + return `adapter '${displayName}' is not available: configured command '${displayCommand}' did not resolve. Check the DAP adapter config for this workspace.`; + } + return ( + ADAPTER_UNAVAILABLE_MESSAGES[adapterName] ?? + `adapter '${displayName}' is not available. Installed adapters: ${getConfiguredAdapters(cwd)}` + ); } async function classifyLaunchProgram(program: string): Promise { @@ -515,7 +543,7 @@ function validateLaunchProgram( if (programKind !== "directory" || adapter.acceptsDirectoryProgram) return; const displayPath = formatPathRelativeToCwd(program, cwd, { trailingSlash: true }); throw new ToolError( - `launch program resolves to a directory: ${displayPath}. Pass an executable file path, or for Python use adapter "debugpy" with program set to the .py file.`, + `launch program resolves to a directory: ${displayPath}. Pass an executable file path or choose an adapter that supports package directories.`, ); } @@ -711,15 +739,16 @@ export class DebugTool implements AgentTool { return cwd; } +interface NestedGoProgram { + moduleRoot: string; + program: string; +} + +async function writeExecutable(filePath: string): Promise { + await fs.mkdir(path.dirname(filePath), { recursive: true }); + await fs.writeFile(filePath, process.platform === "win32" ? "@echo off\r\n" : "#!/bin/sh\n"); + await fs.chmod(filePath, 0o755); +} + +async function writeDlvOverride(cwd: string, command: string): Promise { + await fs.writeFile(path.join(cwd, "dap.json"), JSON.stringify({ adapters: { dlv: { command } } })); +} + +async function setupMissingDlvProject(cwd: string): Promise { + const missingCommand = path.join(cwd, "tools", "missing-dlv"); + await fs.writeFile(path.join(cwd, "go.mod"), "module example.com/app\n\ngo 1.22\n"); + await writeExecutable(path.join(cwd, "bin", "gdb")); + await writeDlvOverride(cwd, missingCommand); + return missingCommand; +} + +async function setupNestedGoProgram(cwd: string): Promise { + const moduleRoot = path.join(cwd, "services", "api"); + const program = path.join(moduleRoot, "main.go"); + await fs.mkdir(moduleRoot, { recursive: true }); + await fs.writeFile(path.join(moduleRoot, "go.mod"), "module example.com/api\n\ngo 1.22\n"); + await fs.writeFile(program, "package main\n\nfunc main() {}\n"); + return { moduleRoot, program }; +} + +function requireSelectedAdapter(selection: LaunchAdapterSelection): DapResolvedAdapter { + if (selection.kind !== "adapter") { + throw new Error(`Expected an available adapter, received '${selection.kind}'`); + } + return selection.adapter; +} + afterEach(async () => { vi.restoreAllMocks(); if (ORIGINAL_OMP_PLUGIN_DIR === undefined) { @@ -63,8 +109,8 @@ describe("DAP adapter configuration", () => { expect(adapter?.launchDefaults).toEqual({ request: "launch", mainClass: "" }); expect(adapter?.attachDefaults).toEqual({ request: "attach", host: "127.0.0.1" }); - const selected = selectLaunchAdapter(path.join("src", "Main.java"), cwd); - expect(selected?.name).toBe("custom-jvm"); + const selected = requireSelectedAdapter(selectLaunchAdapter(path.join("src", "Main.java"), cwd)); + expect(selected.name).toBe("custom-jvm"); }); it("merges partial user overrides over built-in adapters", async () => { @@ -116,9 +162,9 @@ describe("DAP adapter configuration", () => { ].join("\n"), ); - const selected = selectLaunchAdapter("Main.kt", cwd); - expect(selected?.name).toBe("yaml-kotlin"); - expect(selected?.launchDefaults).toEqual({ request: "launch", projectRoot: "." }); + const selected = requireSelectedAdapter(selectLaunchAdapter("Main.kt", cwd)); + expect(selected.name).toBe("yaml-kotlin"); + expect(selected.launchDefaults).toEqual({ request: "launch", projectRoot: "." }); }); it("resolves relative adapter commands from the debug cwd", async () => { @@ -195,4 +241,100 @@ describe("DAP adapter configuration", () => { expect(config["missing-command"]).toBeUndefined(); expect(config.valid?.command).toBe("bun"); }); + + it("reports missing dlv for Go source instead of falling back to a native debugger", async () => { + const cwd = await makeTempDir("omp-dap-go-source-missing-"); + const missingCommand = await setupMissingDlvProject(cwd); + const program = path.join(cwd, "main.go"); + await fs.writeFile(program, "package main\n\nfunc main() {}\n"); + + const selection = selectLaunchAdapter(program, cwd); + + expect(selection).toEqual({ kind: "unavailable", adapterName: "dlv", command: missingCommand }); + }); + + it("reports missing dlv for Go package directories instead of selecting a native debugger", async () => { + const cwd = await makeTempDir("omp-dap-go-directory-missing-"); + const missingCommand = await setupMissingDlvProject(cwd); + const program = path.join(cwd, "cmd", "server"); + await fs.mkdir(program, { recursive: true }); + + const selection = selectLaunchAdapter(program, cwd, undefined, "directory"); + + expect(selection).toEqual({ kind: "unavailable", adapterName: "dlv", command: missingCommand }); + }); + + it("prefers a nested module adapter over cwd and PATH for inferred launches", async () => { + const cwd = await makeTempDir("omp-dap-go-nested-local-"); + const { moduleRoot, program } = await setupNestedGoProgram(cwd); + const nestedDlv = path.join(moduleRoot, "bin", "dlv"); + await writeExecutable(nestedDlv); + await fs.writeFile(path.join(cwd, "go.mod"), "module example.com/repo\n\ngo 1.22\n"); + await writeExecutable(path.join(cwd, "bin", "dlv")); + const whichSpy = vi.spyOn(piUtils, "$which").mockReturnValue(path.join(cwd, "global", "dlv")); + + const selected = requireSelectedAdapter(selectLaunchAdapter(program, cwd)); + + expect(selected.resolvedCommand).toBe(nestedDlv); + expect(whichSpy).not.toHaveBeenCalled(); + }); + + it("uses a nested module adapter when dlv is requested explicitly", async () => { + const cwd = await makeTempDir("omp-dap-go-nested-explicit-"); + const { moduleRoot, program } = await setupNestedGoProgram(cwd); + const nestedDlv = path.join(moduleRoot, "bin", "dlv"); + await writeExecutable(nestedDlv); + const whichSpy = vi.spyOn(piUtils, "$which").mockReturnValue(path.join(cwd, "global", "dlv")); + + const selected = requireSelectedAdapter(selectLaunchAdapter(program, cwd, "dlv")); + + expect(selected.resolvedCommand).toBe(nestedDlv); + expect(whichSpy).not.toHaveBeenCalled(); + }); + + it("prefers the session cwd adapter over PATH after a nested-root miss", async () => { + const cwd = await makeTempDir("omp-dap-go-nested-cwd-"); + const { program } = await setupNestedGoProgram(cwd); + const cwdDlv = path.join(cwd, "bin", "dlv"); + await fs.writeFile(path.join(cwd, "go.mod"), "module example.com/repo\n\ngo 1.22\n"); + await writeExecutable(cwdDlv); + const whichSpy = vi.spyOn(piUtils, "$which").mockReturnValue(path.join(cwd, "global", "dlv")); + + const selected = requireSelectedAdapter(selectLaunchAdapter(program, cwd)); + + expect(selected.resolvedCommand).toBe(cwdDlv); + expect(whichSpy).not.toHaveBeenCalled(); + }); + + it("resolves a local dlv for Go workspaces rooted by go.work", async () => { + const cwd = await makeTempDir("omp-dap-go-work-"); + const program = path.join(cwd, "cmd", "worker"); + const localDlv = path.join(cwd, "bin", "dlv"); + await fs.writeFile(path.join(cwd, "go.work"), "go 1.22\n\nuse ./cmd/worker\n"); + await fs.mkdir(program, { recursive: true }); + await writeExecutable(localDlv); + + const selected = requireSelectedAdapter(selectLaunchAdapter(program, cwd, undefined, "directory")); + + expect(selected.resolvedCommand).toBe(localDlv); + }); + + it("re-resolves an adapter installed after an earlier miss", async () => { + const cwd = await makeTempDir("omp-dap-go-fresh-"); + const program = path.join(cwd, "main.go"); + const command = path.join(cwd, "tools", process.platform === "win32" ? "dlv.cmd" : "dlv"); + await fs.writeFile(path.join(cwd, "go.mod"), "module example.com/cache\n\ngo 1.22\n"); + await fs.writeFile(program, "package main\n\nfunc main() {}\n"); + await writeDlvOverride(cwd, command); + + expect(selectLaunchAdapter(program, cwd)).toEqual({ + kind: "unavailable", + adapterName: "dlv", + command, + }); + + await writeExecutable(command); + const selected = requireSelectedAdapter(selectLaunchAdapter(program, cwd)); + expect(selected.resolvedCommand).toBe(command); + }); }); diff --git a/packages/coding-agent/test/debug/dap-launch-failures.test.ts b/packages/coding-agent/test/debug/dap-launch-failures.test.ts index 6d02e49b7..6e1f63622 100644 --- a/packages/coding-agent/test/debug/dap-launch-failures.test.ts +++ b/packages/coding-agent/test/debug/dap-launch-failures.test.ts @@ -489,7 +489,10 @@ describe("DAP launch failure handling", () => { describe("DebugTool launch validation", () => { it("rejects directory programs when the selected adapter cannot debug a directory", async () => { - const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(TEST_ADAPTER); + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ + kind: "adapter", + adapter: TEST_ADAPTER, + }); try { const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-program-")); try { @@ -523,7 +526,10 @@ describe("DebugTool launch validation", () => { launchDefaults: { request: "launch", mode: "debug", stopOnEntry: true }, acceptsDirectoryProgram: true, }; - const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(dlvAdapter); + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ + kind: "adapter", + adapter: dlvAdapter, + }); const sessionLaunchSpy = spyOn(dapModule.dapSessionManager, "launch").mockImplementation(async opts => { throw Object.assign(new Error("captured launch"), { capturedOptions: opts }); }); @@ -603,7 +609,10 @@ describe("DebugTool launch validation", () => { launchDefaults: { request: "launch", mode: "debug", stopOnEntry: true }, acceptsDirectoryProgram: true, }; - const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(dlvAdapter); + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ + kind: "adapter", + adapter: dlvAdapter, + }); const sessionLaunchSpy = spyOn(dapModule.dapSessionManager, "launch").mockImplementation(async opts => { throw Object.assign(new Error("captured launch"), { capturedOptions: opts }); }); @@ -635,7 +644,11 @@ describe("DebugTool launch validation", () => { }); it("throws targeted 'python not found in PATH' when adapter:'debugpy' is unresolvable for launch", async () => { - const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(null); + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ + kind: "unavailable", + adapterName: "debugpy", + command: "python", + }); try { const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-debugpy-")); try { @@ -685,8 +698,93 @@ describe("DebugTool launch validation", () => { } }); + it("shows the Delve install command when the canonical dlv adapter is unavailable", async () => { + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ + kind: "unavailable", + adapterName: "dlv", + command: "dlv", + }); + try { + const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-hint-")); + try { + await fs.writeFile(path.join(cwd, "main.go"), "package main\n\nfunc main() {}\n"); + const session: ToolSession = { + cwd, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "debug.enabled": true }), + }; + const tool = new DebugTool(session); + + await expect(tool.execute("call", { action: "launch", program: "main.go" })).rejects.toThrow( + /go install github\.com\/go-delve\/delve\/cmd\/dlv@latest/, + ); + } finally { + await removeWithRetries(cwd); + } + } finally { + launchSpy.mockRestore(); + } + }); + + it("points to DAP configuration when a custom adapter command is unavailable", async () => { + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ + kind: "unavailable", + adapterName: "dlv", + command: "./bin/missing-dlv", + }); + try { + const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-config-")); + try { + await fs.writeFile(path.join(cwd, "main.go"), "package main\n\nfunc main() {}\n"); + const session: ToolSession = { + cwd, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "debug.enabled": true }), + }; + const tool = new DebugTool(session); + + await expect(tool.execute("call", { action: "launch", program: "main.go" })).rejects.toThrow( + /configured command '\.\/bin\/missing-dlv' did not resolve.*DAP adapter config/, + ); + } finally { + await removeWithRetries(cwd); + } + } finally { + launchSpy.mockRestore(); + } + }); + + it("shows the rdbg install command for explicit Ruby attach", async () => { + const attachSpy = spyOn(dapModule, "selectAttachAdapter").mockReturnValue(null); + try { + const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-rdbg-attach-")); + try { + const session: ToolSession = { + cwd, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "debug.enabled": true }), + }; + const tool = new DebugTool(session); + + await expect(tool.execute("call", { action: "attach", pid: 1234, adapter: "rdbg" })).rejects.toThrow( + /gem install debug/, + ); + } finally { + await removeWithRetries(cwd); + } + } finally { + attachSpy.mockRestore(); + } + }); + it("falls back to the generic 'No debugger adapter' error when adapter is unspecified", async () => { - const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(null); + const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue({ kind: "none" }); try { const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-noadapter-")); try { From d0a4bc47738977050ebe1f4a07b1a2f7d3963a02 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 14:07:22 +0200 Subject: [PATCH 071/205] fix(ai): transitioned reasoning summaries to incremental append-only deltas - Introduced `SequentialCutoffSummaryState` to track reasoning summaries globally across all response items. - Updated `CodexStreamRuntime` to maintain cumulative summary state instead of per-item snapshots. - Modified `applyReasoningSummaryDone` and `finalizeReasoningThinking` to emit only append-only text deltas, suppressing replayed content from previous items. - Added comprehensive test coverage to verify correct incremental thinking output across multiple sequential reasoning items. --- packages/ai/CHANGELOG.md | 1 + .../src/providers/openai-codex-responses.ts | 15 +- packages/ai/src/providers/openai-shared.ts | 93 +++++++--- .../test/openai-codex-responses-lite.test.ts | 169 ++++++++++++++++++ 4 files changed, 249 insertions(+), 29 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7478d451b..a96424080 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -20,6 +20,7 @@ - Fixed OpenAI-compatible chat-completions streams to preserve vLLM-style trailing cached-token usage chunks, ensuring accurate cacheRead and billable input session statistics. - Fixed xai-oauth/grok-4.5 Responses requests to omit the unsupported reasoning.summary field while preserving the reasoning.effort payload. - Fixed Codex OAuth credential selection to re-check blocked accounts during ranking and clear stale usage-limit blocks once live usage indicates recovery. +- Fixed sequential-cutoff reasoning summaries duplicating section headers across Codex reasoning items by tracking the cumulative summary response-globally, so replayed sections and replay-only items no longer re-emit text earlier thinking blocks already streamed. ## [16.3.15] - 2026-07-09 diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index a96d5b2a9..0fa0f720c 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -95,6 +95,7 @@ import { buildResponsesDeltaInput, convertResponsesAssistantMessage, convertResponsesInputContent, + createSequentialCutoffSummaryState, encodeResponsesToolCallId, encodeTextSignatureV1, finalizeCustomToolCallInputDone, @@ -106,6 +107,7 @@ import { normalizeOpenAIPromptCacheKey, populateResponsesUsageFromResponse, promoteResponsesToolUseStopReason, + type SequentialCutoffSummaryState, } from "./openai-shared"; import { transformMessages } from "./transform-messages"; @@ -685,6 +687,8 @@ class CodexStreamRuntime { currentItem: CodexEventItem | null = null; currentBlock: CodexOutputBlock | null = null; nativeOutputItems: Array> = []; + /** Sequential-cutoff summary sections/emitted text, global to the response (indices span reasoning items). */ + cutoffSummaries: SequentialCutoffSummaryState = createSequentialCutoffSummaryState(); websocketStreamRetries = 0; providerRetryAttempt = 0; sawTerminalEvent = false; @@ -717,6 +721,7 @@ class CodexStreamRuntime { this.currentItem = null; this.currentBlock = null; this.nativeOutputItems.length = 0; + this.cutoffSummaries = createSequentialCutoffSummaryState(); } /** @@ -1788,7 +1793,7 @@ class CodexStreamProcessor { ? Math.trunc(rawEvent.summary_index) : 0; applyReasoningSummaryDone( - entry.item, + this.runtime.cutoffSummaries, entry.block, typeof rawEvent.text === "string" ? rawEvent.text : "", summaryIndex, @@ -1937,9 +1942,11 @@ class CodexStreamProcessor { const contentIndex = entry?.contentIndex ?? output.content.length - 1; if (item.type === "reasoning" && block?.type === "thinking") { - block.thinking = finalizeReasoningThinking(item, block.thinking, { - cumulativeSummarySnapshots: this.#sequentialCutoffSummaries, - }); + block.thinking = finalizeReasoningThinking( + item, + block.thinking, + this.#sequentialCutoffSummaries ? this.runtime.cutoffSummaries : undefined, + ); block.thinkingSignature = JSON.stringify(item); stream.push({ type: "thinking_end", diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 47a803349..f70989560 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1702,6 +1702,26 @@ export function appendReasoningSummaryPart( item.summary.push(part); } +/** + * Response-global accumulator for the sequential-cutoff summary contract. + * + * Summary indices are cumulative across ALL reasoning items in a response: + * each new reasoning item replays the previous item's last completed section + * (`.done` at index N-1) before streaming its own, and replay-only items may + * add nothing new. Folding per item would re-emit every replayed section, so + * the canonical summary and the emitted text span items and live here. + */ +export interface SequentialCutoffSummaryState { + /** Latest full text per response-global summary index. */ + summary: ResponseReasoningItem["summary"]; + /** Canonical summary text already emitted as thinking deltas across all blocks. */ + emitted: string; +} + +export function createSequentialCutoffSummaryState(): SequentialCutoffSummaryState { + return { summary: [], emitted: "" }; +} + // Sequential-cutoff streams may repeat the full canonical summary as later parts. function foldReasoningSummary(parts: ResponseReasoningItem["summary"] | undefined): string { if (!parts) return ""; @@ -1719,24 +1739,41 @@ function foldReasoningSummary(parts: ResponseReasoningItem["summary"] | undefine export function finalizeReasoningThinking( item: ResponseReasoningItem, streamedThinking: string, - options: { cumulativeSummarySnapshots?: boolean } = {}, + cutoff?: SequentialCutoffSummaryState, ): string { - const summaryThinking = options.cumulativeSummarySnapshots - ? foldReasoningSummary(item.summary) - : (item.summary?.map(part => part.text).join("\n\n") ?? ""); - if ( - options.cumulativeSummarySnapshots && - streamedThinking && - summaryThinking && - summaryThinking !== streamedThinking - ) { - return streamedThinking; - } + if (cutoff) return finalizeCutoffReasoningThinking(item, streamedThinking, cutoff); + const summaryThinking = item.summary?.map(part => part.text).join("\n\n") ?? ""; if (summaryThinking) return summaryThinking; const contentThinking = item.content?.[0]?.type === "reasoning_text" ? (item.content[0].text ?? "") : ""; return contentThinking || streamedThinking || ""; } +function finalizeCutoffReasoningThinking( + item: ResponseReasoningItem, + streamedThinking: string, + cutoff: SequentialCutoffSummaryState, +): string { + // The block's streamed deltas are authoritative: final text must never + // disagree with what delta consumers already rendered. + if (streamedThinking) return streamedThinking; + const summaryThinking = foldReasoningSummary(item.summary); + if (summaryThinking) { + // The done payload carries the response-cumulative summary. Emit only + // what no earlier block already emitted; replay-only items finalize empty. + if (cutoff.emitted.startsWith(summaryThinking)) return ""; + if (!cutoff.emitted || summaryThinking.startsWith(cutoff.emitted)) { + const suffix = summaryThinking.slice(cutoff.emitted.length).replace(/^\n+/, ""); + // Adopt the payload as canonical so later items cannot replay this text. + cutoff.summary = item.summary?.map(part => ({ ...part })) ?? []; + cutoff.emitted = summaryThinking; + return suffix; + } + // Diverged from streamed text — the deltas already shown win. + return ""; + } + return item.content?.[0]?.type === "reasoning_text" ? (item.content[0].text ?? "") : ""; +} + export function appendReasoningSummaryTextDelta( item: ResponseReasoningItem, block: ThinkingContent, @@ -1771,13 +1808,15 @@ export function appendReasoningSummaryPartDone( /** * Applies an atomic `response.reasoning_summary_text.done` snapshot. * - * Sequential-cutoff streams can replay an index or send the accumulated - * summary as a later part. Rebuild the canonical summary and emit only its - * append-only suffix. Divergent corrections stay buffered until finalization - * so delta consumers never receive suffixes based on unseen replacement text. + * Sequential-cutoff summary indices are response-global: later reasoning items + * replay earlier sections, resend the accumulated summary as one part, or + * complete without new sections. The canonical summary is rebuilt in `state` + * (spanning items) and only its append-only suffix is emitted into the current + * block. Divergent corrections stay buffered until finalization so delta + * consumers never receive suffixes based on unseen replacement text. */ export function applyReasoningSummaryDone( - item: ResponseReasoningItem, + state: SequentialCutoffSummaryState, block: ThinkingContent, text: string, summaryIndex: number, @@ -1785,16 +1824,20 @@ export function applyReasoningSummaryDone( output: AssistantMessage, contentIndex: number, ): void { - item.summary = item.summary || []; - while (item.summary.length <= summaryIndex) { - item.summary.push({ type: "summary_text", text: "" }); + while (state.summary.length <= summaryIndex) { + state.summary.push({ type: "summary_text", text: "" }); } - item.summary[summaryIndex].text = text; - const after = foldReasoningSummary(item.summary); - if (!after.startsWith(block.thinking)) return; - const delta = after.slice(block.thinking.length); + state.summary[summaryIndex].text = text; + const after = foldReasoningSummary(state.summary); + if (!after.startsWith(state.emitted)) return; + let delta = after.slice(state.emitted.length); if (!delta) return; - block.thinking = after; + state.emitted = after; + // A fresh block starts a new section: drop the inter-section separator so + // each thinking block stands alone. + if (!block.thinking) delta = delta.replace(/^\n+/, ""); + if (!delta) return; + block.thinking += delta; stream.push({ type: "thinking_delta", contentIndex, delta, partial: output }); } diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 169f578a9..992ee26d3 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -897,4 +897,173 @@ describe("openai-codex concurrent reasoning summaries", () => { const text = result.content.find(block => block.type === "text"); expect(text?.text).toBe("Hello"); }); + + it("does not replay earlier sections across reasoning items under sequential cutoff", async () => { + // Real gpt-5.6 sessions send response-GLOBAL summary indices: each new + // reasoning item replays the previous item's last completed section + // (`.done` at index N-1) before streaming its own, replay-only items add + // nothing, and every `output_item.done` payload carries the cumulative + // summary array. Folding per item duplicated every section header. + const model = createCodexModel("gpt-5.6-terra"); + const events: Array> = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.reasoning_summary_text.done", + item_id: "rs_1", + output_index: 0, + summary_index: 0, + text: "Planning refactor", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [{ type: "summary_text", text: "Planning refactor" }] }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "reasoning", id: "rs_2", summary: [] }, + }, + // Replay of the previous item's section, then the new one. + { + type: "response.reasoning_summary_text.done", + item_id: "rs_2", + output_index: 1, + summary_index: 0, + text: "Planning refactor", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "rs_2", + output_index: 1, + summary_index: 1, + text: "Designing resolution", + }, + { + type: "response.output_item.done", + output_index: 1, + item: { + type: "reasoning", + id: "rs_2", + summary: [ + { type: "summary_text", text: "Planning refactor" }, + { type: "summary_text", text: "Designing resolution" }, + ], + }, + }, + { + type: "response.output_item.added", + output_index: 2, + item: { type: "reasoning", id: "rs_3", summary: [] }, + }, + // Replay-only item: no new section arrives before it closes. + { + type: "response.reasoning_summary_text.done", + item_id: "rs_3", + output_index: 2, + summary_index: 1, + text: "Designing resolution", + }, + { + type: "response.output_item.done", + output_index: 2, + item: { + type: "reasoning", + id: "rs_3", + summary: [ + { type: "summary_text", text: "Planning refactor" }, + { type: "summary_text", text: "Designing resolution" }, + ], + }, + }, + { + type: "response.output_item.added", + output_index: 3, + item: { type: "reasoning", id: "rs_4", summary: [] }, + }, + // Payload-only item: its new section never streams a `.done` event. + { + type: "response.output_item.done", + output_index: 3, + item: { + type: "reasoning", + id: "rs_4", + summary: [ + { type: "summary_text", text: "Planning refactor" }, + { type: "summary_text", text: "Designing resolution" }, + { type: "summary_text", text: "Enhancing caching" }, + ], + }, + }, + { + type: "response.output_item.added", + output_index: 4, + item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", item_id: "msg_1", output_index: 4, delta: "Hello" }, + { + type: "response.output_item.done", + output_index: 4, + item: { + type: "message", + id: "msg_1", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "Hello" }], + }, + }, + { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + const fetchMock = createCodexFetchMock(createCodexSse(events), () => {}); + + const stream = streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + reasoning: "medium", + }); + const deltasByBlock = new Map(); + for await (const event of stream) { + if (event.type === "thinking_delta") { + deltasByBlock.set(event.contentIndex, (deltasByBlock.get(event.contentIndex) ?? "") + event.delta); + } + } + const result = await stream.result(); + + const thinkingBlocks = result.content.filter(block => block.type === "thinking"); + expect(thinkingBlocks.map(block => block.thinking)).toEqual([ + "Planning refactor", + "Designing resolution", + "", + "Enhancing caching", + ]); + // Streamed deltas match each block that streamed; the payload-only block + // surfaces its unseen suffix at finalization without a delta. + expect([...deltasByBlock.entries()]).toEqual([ + [0, "Planning refactor"], + [1, "Designing resolution"], + ]); + // The replay-only block keeps its signed reasoning item so history replay + // still round-trips encrypted reasoning. + const replayOnly = thinkingBlocks[2]; + expect(replayOnly?.thinkingSignature).toBeDefined(); + expect(JSON.parse(replayOnly?.thinkingSignature ?? "{}").id).toBe("rs_3"); + const text = result.content.find(block => block.type === "text"); + expect(text?.text).toBe("Hello"); + }); }); From 4bae9a42ab764336ab17fdf321fe8d23e8ed8fcd Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 12:08:43 +0000 Subject: [PATCH 072/205] fix(providers): required copilot vision confirmation Keep non-personal Copilot endpoints text-only when discovery omits supports.vision, while preserving explicit vision support. Fixes #4779 --- .../src/provider-models/openai-compat.ts | 17 ++++++---- .../test/github-copilot-model-limits.test.ts | 34 +++++++++++++++++++ 2 files changed, 45 insertions(+), 6 deletions(-) diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index b4a7ba007..2f52dbe6f 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -16,7 +16,12 @@ import { getBundledModels } from "../models"; import type { Api, FetchImpl, Model, ModelSpec, OpenAICompat, Provider, ThinkingConfig } from "../types"; import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils"; import { coreWeaveProjectHeaders } from "../wire/coreweave"; -import { COPILOT_API_HEADERS, getGitHubCopilotBaseUrl, parseGitHubCopilotApiKey } from "../wire/github-copilot"; +import { + COPILOT_API_HEADERS, + getGitHubCopilotBaseUrl, + isPersonalGitHubCopilotBaseUrl, + parseGitHubCopilotApiKey, +} from "../wire/github-copilot"; import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references"; const MODELS_DEV_URL = "https://models.dev/api.json"; @@ -3692,11 +3697,11 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana const api = inferCopilotApi(defaults.id); const supportsVision = extractCopilotSupportsVision(entry); const input: ModelSpec["input"] = - supportsVision === undefined - ? (reference?.input ?? defaults.input) - : supportsVision - ? ["text", "image"] - : ["text"]; + supportsVision === true + ? ["text", "image"] + : supportsVision === false || !isPersonalGitHubCopilotBaseUrl(baseUrl) + ? ["text"] + : (reference?.input ?? defaults.input); // With COPILOT_API_HEADERS the served window is the long-context // ceiling; the default tier ends at token_prices.default.context_max // prompt tokens. Cap the base entry to the default tier — the long diff --git a/packages/catalog/test/github-copilot-model-limits.test.ts b/packages/catalog/test/github-copilot-model-limits.test.ts index b6ecfe3d4..aa5cc66e7 100644 --- a/packages/catalog/test/github-copilot-model-limits.test.ts +++ b/packages/catalog/test/github-copilot-model-limits.test.ts @@ -605,6 +605,40 @@ describe("github copilot vision endpoint policy", () => { } }); + it("maps omitted upstream vision to text-only on non-personal Copilot endpoints", async () => { + for (const endpoint of [ + { + apiKey: businessApiKey, + baseUrl: "https://api.business.githubcopilot.com", + token: "ghu_business_token", + }, + { + apiKey: enterpriseApiKey, + baseUrl: "https://copilot-api.ghe.example.com", + token: "ghu_enterprise_token", + }, + ]) { + const { models } = await discoverCopilotModels( + { + data: [ + tieredCopilotEntry({ + id: "claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + window: 200_000, + maxOutput: 32_000, + }), + ], + }, + endpoint.apiKey, + endpoint.baseUrl, + endpoint.token, + ); + const model = models.find(candidate => candidate.id === "claude-sonnet-4.6"); + expect(model?.baseUrl).toBe(endpoint.baseUrl); + expect(model?.input).toEqual(["text"]); + } + }); + it("keeps vision on the canonical personal Copilot endpoint", async () => { const { models } = await discoverCopilotModels({ data: [ From 68bc6ba605f2005af9c5fa6e3e45947060001821 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 14:10:41 +0200 Subject: [PATCH 073/205] test(coding-agent): updated thinking model metadata expectations - Update Ollama discovery tests to reflect the wire effort vocabulary. - Adjust model registry test expectations to treat adaptive effort ladders as verbatim, removing the legacy effortMap backfilling behavior. --- packages/coding-agent/test/model-discovery.test.ts | 4 ++-- .../test/model-registry-runtime-provider.test.ts | 5 ++--- packages/coding-agent/test/model-registry.test.ts | 11 ++++------- 3 files changed, 8 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index 393e60645..d700be529 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -700,8 +700,8 @@ describe("ModelRegistry runtime discovery", () => { expect(qwen?.reasoning).toBe(true); expect(qwen?.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High], - effortMap: { [Effort.Minimal]: Effort.Low }, + // Local Ollama's wire effort vocabulary is low/medium/high/max. + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], }); const llama = registry.find("ollama", "llama3.2:3b"); diff --git a/packages/coding-agent/test/model-registry-runtime-provider.test.ts b/packages/coding-agent/test/model-registry-runtime-provider.test.ts index 977e8a665..35aada7e4 100644 --- a/packages/coding-agent/test/model-registry-runtime-provider.test.ts +++ b/packages/coding-agent/test/model-registry-runtime-provider.test.ts @@ -371,9 +371,8 @@ describe("ModelRegistry runtime provider registration", () => { expect(model?.thinking).toEqual({ mode: "anthropic-adaptive", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High], - // Wire facts are backfilled from identity; non-claude ids get the - // 4-tier adaptive map, filtered to the declared efforts (no xhigh). - effortMap: { minimal: "low" }, + // Adaptive ladders are wire-exact (no backfilled effortMap); only + // requiresEffort is backfilled from identity. requiresEffort: true, }); }); diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 02bbfc5e3..e10993c40 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -1095,14 +1095,11 @@ describe("ModelRegistry", () => { }); }); - test("custom models preserve explicit thinking and gain backfilled wire facts", () => { + test("custom models preserve explicit thinking verbatim", () => { const model = getModelsForProvider(thinkingCustom, "anthropic").find(m => m.id === "claude-custom"); - expect(model?.thinking).toEqual({ - ...customThinking, - // Versionless claude ids resolve to the 4-tier adaptive wire map, - // filtered to the declared efforts (no xhigh). - effortMap: { minimal: "low" }, - }); + // Adaptive effort ladders are wire-exact — explicit thinking passes + // through without a backfilled effortMap. + expect(model?.thinking).toEqual(customThinking); }); test("model overrides can replace canonical thinking metadata", () => { From a0f7266fbc623817fdf11354f99bb0afd0d7bb6c Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 14:11:05 +0200 Subject: [PATCH 074/205] chore: bump version to 16.4.0 --- Cargo.lock | 10 ++--- Cargo.toml | 2 +- bun.lock | 62 +++++++++++++-------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 24 +++++------ packages/agent/CHANGELOG.md | 2 + packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 + packages/ai/package.json | 2 +- packages/catalog/CHANGELOG.md | 2 + packages/catalog/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 + packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 25 files changed, 75 insertions(+), 65 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 125f5b658..2fdf19aee 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2881,7 +2881,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.3.15" +version = "16.4.0" dependencies = [ "anyhow", "ast-grep-core", @@ -2950,7 +2950,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.3.15" +version = "16.4.0" dependencies = [ "async-trait", "libc", @@ -2962,7 +2962,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.3.15" +version = "16.4.0" dependencies = [ "anyhow", "arboard", @@ -3015,7 +3015,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.3.15" +version = "16.4.0" dependencies = [ "anyhow", "brush-builtins", @@ -3064,7 +3064,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.3.15" +version = "16.4.0" dependencies = [ "dashmap", "globset", diff --git a/Cargo.toml b/Cargo.toml index 652fd749f..0e637b62a 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.3.15" +version = "16.4.0" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index d179b61c4..0cae7fb8e 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.15", + "version": "16.4.0", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.15", + "version": "16.4.0", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.3.15", + "version": "16.4.0", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.3.15", + "version": "16.4.0", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.3.15", + "version": "16.4.0", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.3.15", + "version": "16.4.0", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.3.15", + "version": "16.4.0", "devDependencies": { "@types/bun": "catalog:", }, @@ -338,18 +338,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.15", - "@oh-my-pi/omp-stats": "16.3.15", - "@oh-my-pi/pi-agent-core": "16.3.15", - "@oh-my-pi/pi-ai": "16.3.15", - "@oh-my-pi/pi-catalog": "16.3.15", - "@oh-my-pi/pi-coding-agent": "16.3.15", - "@oh-my-pi/pi-mnemopi": "16.3.15", - "@oh-my-pi/pi-natives": "16.3.15", - "@oh-my-pi/pi-tui": "16.3.15", - "@oh-my-pi/pi-utils": "16.3.15", - "@oh-my-pi/pi-wire": "16.3.15", - "@oh-my-pi/snapcompact": "16.3.15", + "@oh-my-pi/hashline": "16.4.0", + "@oh-my-pi/omp-stats": "16.4.0", + "@oh-my-pi/pi-agent-core": "16.4.0", + "@oh-my-pi/pi-ai": "16.4.0", + "@oh-my-pi/pi-catalog": "16.4.0", + "@oh-my-pi/pi-coding-agent": "16.4.0", + "@oh-my-pi/pi-mnemopi": "16.4.0", + "@oh-my-pi/pi-natives": "16.4.0", + "@oh-my-pi/pi-tui": "16.4.0", + "@oh-my-pi/pi-utils": "16.4.0", + "@oh-my-pi/pi-wire": "16.4.0", + "@oh-my-pi/snapcompact": "16.4.0", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -967,7 +967,7 @@ "bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="], - "caniuse-lite": ["caniuse-lite@1.0.30001800", "", {}, "sha512-MMHtuAz9Ys840zAY5F4k6fV5GaivZ9sPk+nz0mY+GYVzRBnYkN0mpqkSR92oWRQ19yQWo4HvBV/FnC16AJX8MA=="], + "caniuse-lite": ["caniuse-lite@1.0.30001803", "", {}, "sha512-g/uHREV2ZpK9qMalCsWaxmA6ol+DX8GYhuf3T40RKoP+oL7vhRJh8LNt73PCjpnR6l14FzfPrB5Yux4PKm2meg=="], "chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="], @@ -1045,7 +1045,7 @@ "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], - "electron-to-chromium": ["electron-to-chromium@1.5.387", "", {}, "sha512-TaxwufTFDufvPEoXdhwVrA3UdFWBeWGkYoJ1K8ldF1xe6gKfth6iRNS5lTQ5JPNOHdGQm8PT1QYKUqFLCiUefQ=="], + "electron-to-chromium": ["electron-to-chromium@1.5.388", "", {}, "sha512-Pl/aJaqOOxYxda3vcx1IKSJimwYXHDkEnGn0F+kG2EE68dDtx2uCinaS+Vih8Z91B9t8CSAbiF/HKyWcnXjhzw=="], "emnapi": ["emnapi@1.11.2", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-iMt/XQc69fFn2EvcU6tm14HmXKwyy0lnABugsQlqp6xFuZIUuO+ONVSg2mz+MTVF8WbC+bic65AvRXdoldALKg=="], @@ -1341,17 +1341,17 @@ "sherpa-onnx-darwin-arm64": ["sherpa-onnx-darwin-arm64@1.13.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-9x86Cbf+BDFONdtCPM3cnjvtAW0ER8tMaHK5pVfz+SHPt8GeuwRXaiR/BzcByFBUyxCgmceO09/WMZOCi44P/g=="], - "sherpa-onnx-darwin-x64": ["sherpa-onnx-darwin-x64@1.13.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-TVQ35g7JIpDPB1lUDdcog+JtI0cI45ZzOnvHXm0DtWs/dgxnJXtWMY3uLRtBbLnysV9j5ljffwZ1IX9VDHsCzQ=="], + "sherpa-onnx-darwin-x64": ["sherpa-onnx-darwin-x64@1.13.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-6RGeis9K9gV/UQWOgd6Rf3iqXr2/YsBQswxHaCR4hrYkHfEIpHMfFmRWLt6nJJCOWgYW2xFxEd9yzjrafAV/Pw=="], "sherpa-onnx-linux-arm64": ["sherpa-onnx-linux-arm64@1.13.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-uDtZkkoP6QQ/3DHOscCpEZ2WpaiHUQsDpbyYaHURrJ7DbsjqGnS6G8l+R589Ro5Bf282QElzBy3okwxXbt3Kxw=="], - "sherpa-onnx-linux-x64": ["sherpa-onnx-linux-x64@1.13.3", "", { "os": "linux", "cpu": "x64" }, "sha512-OFVK0GYwKwKNsjxbPmfcLQm/dfA0IwAoiIQJ96s+eFYcDqhlapcY06ocdb7SNluGBcM7xgU5jEW2QXBkMIOEvQ=="], + "sherpa-onnx-linux-x64": ["sherpa-onnx-linux-x64@1.13.4", "", { "os": "linux", "cpu": "x64" }, "sha512-WZh5NCkGPFHHpYSd78iN4OnmxQeSTGyt9uZskH+im/NFHQ7elQ7B0sLzCMeRpvJxiIKvd9C6WxIJ4hYaxClfsQ=="], "sherpa-onnx-node": ["sherpa-onnx-node@1.13.2", "", { "optionalDependencies": { "sherpa-onnx-darwin-arm64": "^1.13.2", "sherpa-onnx-darwin-x64": "^1.13.2", "sherpa-onnx-linux-arm64": "^1.13.2", "sherpa-onnx-linux-x64": "^1.13.2", "sherpa-onnx-win-ia32": "^1.13.2", "sherpa-onnx-win-x64": "^1.13.2" } }, "sha512-uIH6SA5Or4pb8HlCYWB3K54XkMtzdef4/tkw1amtIf8GB1tt6hQLpur9p2jSFNfTYRyzZ8XrXofxefXQ0A7EUA=="], - "sherpa-onnx-win-ia32": ["sherpa-onnx-win-ia32@1.13.3", "", { "os": "win32", "cpu": "ia32" }, "sha512-VDZh1M7Ccx/bkP3WwBCFoJzwAwq+b5nR1KRkYRz5p1w5bfhzfa3ACBGr7vpUt5AGUge4qSLe0MSKXyKtSmy1uA=="], + "sherpa-onnx-win-ia32": ["sherpa-onnx-win-ia32@1.13.4", "", { "os": "win32", "cpu": "ia32" }, "sha512-/JbPjldrfNv+t+uIS3MlkuhfIf5l3FHUGkRC2oRXgjRqOaVmEyP3vLlQ7dTa4J7raG5oB8c3GoPjuSWSqT9GOQ=="], - "sherpa-onnx-win-x64": ["sherpa-onnx-win-x64@1.13.3", "", { "os": "win32", "cpu": "x64" }, "sha512-ZQzcSmFvZK4jzmtWckqxocDUuEjYnBV2MHrDD21HPTeUMfGdE9yfvuSPpesIVfdzKbQzIQY42RAcfZEGWu0FbQ=="], + "sherpa-onnx-win-x64": ["sherpa-onnx-win-x64@1.13.4", "", { "os": "win32", "cpu": "x64" }, "sha512-R0PWby1VxC14TDZPq7GcfSyXSY6SAFO8Y4JwdCdqouFmeXkZ1L7Is9m98C9KxQ0dN7ZtDzhAmE/43FUs/elXRQ=="], "signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 5b93f4516..0ffc82553 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_3_15")] +#[napi(js_name = "__piNativesV16_4_0")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index a63330c5c..0d448221b 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.15", - "@oh-my-pi/omp-stats": "16.3.15", - "@oh-my-pi/pi-agent-core": "16.3.15", - "@oh-my-pi/pi-ai": "16.3.15", - "@oh-my-pi/pi-catalog": "16.3.15", - "@oh-my-pi/pi-coding-agent": "16.3.15", - "@oh-my-pi/pi-mnemopi": "16.3.15", - "@oh-my-pi/pi-natives": "16.3.15", - "@oh-my-pi/pi-tui": "16.3.15", - "@oh-my-pi/pi-utils": "16.3.15", - "@oh-my-pi/pi-wire": "16.3.15", - "@oh-my-pi/snapcompact": "16.3.15", + "@oh-my-pi/hashline": "16.4.0", + "@oh-my-pi/omp-stats": "16.4.0", + "@oh-my-pi/pi-agent-core": "16.4.0", + "@oh-my-pi/pi-ai": "16.4.0", + "@oh-my-pi/pi-catalog": "16.4.0", + "@oh-my-pi/pi-coding-agent": "16.4.0", + "@oh-my-pi/pi-mnemopi": "16.4.0", + "@oh-my-pi/pi-natives": "16.4.0", + "@oh-my-pi/pi-tui": "16.4.0", + "@oh-my-pi/pi-utils": "16.4.0", + "@oh-my-pi/pi-wire": "16.4.0", + "@oh-my-pi/snapcompact": "16.4.0", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index b5ebcd148..972e4f3a4 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.0] - 2026-07-10 + ### Added - Added the `ThinkingLevel.Max` ("max") configuration option, mapping to the `Effort.Max` tier for supported models. diff --git a/packages/agent/package.json b/packages/agent/package.json index 0e5b1961c..d87c298a5 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.15", + "version": "16.4.0", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index a96424080..eda640539 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.0] - 2026-07-10 + ### Added - Added "max" as a first-class reasoning effort option across providers (including Anthropic, Google, Bedrock, and OpenAI), supporting a maximum reasoning budget of 32,768 tokens. diff --git a/packages/ai/package.json b/packages/ai/package.json index 1124f3f13..634515350 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.3.15", + "version": "16.4.0", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 3c91836b9..bc7d74949 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.0] - 2026-07-10 + ### Breaking Changes - Redesigned reasoning effort ladders to be wire-exact, removing the shifted five-tier effort mapping. Models now expose exactly the effort tiers their upstream APIs accept, mapped 1:1. Removed SHIFTED_FIVE_TIER_EFFORT_MAP, ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER, and per-host xhigh-to-max alias maps. Selecting an unsupported tier now automatically clamps down via clampThinkingLevelForModel. Devin effort routing is now mapped 1:1 onto per-tier siblings. diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 653aac2e6..1ad9e3356 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.3.15", + "version": "16.4.0", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 15ad4eaa6..04ae4eb4d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.0] - 2026-07-10 + ### Breaking Changes - Renamed the bundled agent explore to scout, including its configuration keys, prompt files, and task definitions. Any configurations, allowlists, or invocations referencing explore must now use scout. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index c66b13e2f..d64ab8d54 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.15", + "version": "16.4.0", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 73c31c71a..74679635f 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.3.15", + "version": "16.4.0", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index b5a62f06e..a6a86829d 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.15", + "version": "16.4.0", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index bb54f9bf5..4da2cdc32 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -170,7 +170,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_3_15(): void +export declare function __piNativesV16_4_0(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index f439ff2c7..3b6b45f4b 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_3_15 = nativeBindings.__piNativesV16_3_15; +export const __piNativesV16_4_0 = nativeBindings.__piNativesV16_4_0; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 183a3876d..657ac54bd 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.3.15", + "version": "16.4.0", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index 88b2c10be..a2df06664 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.3.15", + "version": "16.4.0", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index 329574cb8..faa1c8032 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.3.15", + "version": "16.4.0", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 0e71287c1..693be1114 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.3.15", + "version": "16.4.0", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 149e4c8a5..e794cb8c5 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.0] - 2026-07-10 + ### Fixed - Fixed terminal flickering during session resume, replacement, or resizing on terminals that do not support synchronized output. diff --git a/packages/tui/package.json b/packages/tui/package.json index fea5e808d..02a414325 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.3.15", + "version": "16.4.0", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 2e186b008..1bbeb5e2f 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.3.15", + "version": "16.4.0", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index 51c90f13f..e57be9695 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.3.15", + "version": "16.4.0", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From e58d2c460cbec43fc8b13511079abaa2c1bf443c Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 12:20:39 +0000 Subject: [PATCH 075/205] fix(providers): preserved copilot vision denials Treat GitHub Copilot discovery input as authoritative during model merges so explicit supports.vision=false is not OR-upgraded from bundled references. Fixes #4779 --- packages/catalog/src/model-manager.ts | 11 +++-- .../test/github-copilot-model-limits.test.ts | 41 +++++++++++++++++++ 2 files changed, 48 insertions(+), 4 deletions(-) diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index eccec451f..f40a5f639 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -351,11 +351,14 @@ function fingerprintStatic( function mergeDynamicModel(existingModel: Model, dynamicModel: Model): Model { // When discovery resolves the same model id to a different endpoint (e.g. // a GitHub Copilot business/enterprise host), the bundled reference's - // capabilities are pinned to the canonical host and no longer apply — - // honour the dynamic value alone. Same-endpoint merges still OR-upgrade so - // a discovery that omits the capability flag doesn't drop bundled vision. + // capabilities are pinned to another endpoint and no longer apply. Copilot + // dynamic discovery also pre-applies the correct image fallback for omitted + // `supports.vision`, so its explicit `false` must not be OR-upgraded by the + // canonical bundled model. const endpointChanged = existingModel.baseUrl !== dynamicModel.baseUrl; - const supportsImage = endpointChanged + const dynamicInputAuthoritative = + endpointChanged || (existingModel.provider === "github-copilot" && dynamicModel.provider === "github-copilot"); + const supportsImage = dynamicInputAuthoritative ? dynamicModel.input.includes("image") : existingModel.input.includes("image") || dynamicModel.input.includes("image"); // Re-build from spec stage: sparse compat comes from `compatConfig` (the diff --git a/packages/catalog/test/github-copilot-model-limits.test.ts b/packages/catalog/test/github-copilot-model-limits.test.ts index aa5cc66e7..c21802a5d 100644 --- a/packages/catalog/test/github-copilot-model-limits.test.ts +++ b/packages/catalog/test/github-copilot-model-limits.test.ts @@ -656,6 +656,47 @@ describe("github copilot vision endpoint policy", () => { expect(model?.input).toEqual(["text", "image"]); }); + it("keeps explicit upstream vision false text-only through the personal endpoint manager merge", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-copilot-vision-")); + try { + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const url = typeof input === "string" ? input : input.toString(); + expect(url).toBe("https://api.githubcopilot.com/models"); + expect(init?.method).toBe("GET"); + expect(getHeaderValue(init?.headers, "Authorization")).toBe("Bearer copilot-test-key"); + return new Response( + JSON.stringify({ + data: [ + tieredCopilotEntry({ + id: "claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + window: 200_000, + maxOutput: 32_000, + vision: false, + }), + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }); + + const bundled = getBundledModel("github-copilot", "claude-sonnet-4.6"); + expect(bundled?.input).toEqual(["text", "image"]); + + const options = githubCopilotModelManagerOptions({ apiKey: "copilot-test-key", fetch: fetchMock }); + const manager = createModelManager({ + ...options, + cacheDbPath: path.join(tempDir, "models.db"), + }); + const { models } = await manager.refresh("online"); + const model = models.find(candidate => candidate.id === "claude-sonnet-4.6"); + expect(model?.baseUrl).toBe("https://api.githubcopilot.com"); + expect(model?.input).toEqual(["text"]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("keeps the merged Model image-capable when business discovery confirms a vision-capable bundled reference", async () => { // Bundled `claude-sonnet-4.6` ships with `input=['text','image']`. // Discovery against the business host confirms the same upstream vision From b35e4c41326cde171624b3795fa9b5aabd51a668 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 12:28:51 +0000 Subject: [PATCH 076/205] fix(tui): honored move overlay width Rendered the /move directory picker using the overlay layout width instead of the legacy fixed 68-column frame. Added regression coverage for non-68-column overlay frames. Fixes #5067 --- packages/coding-agent/CHANGELOG.md | 1 + .../components/__tests__/move-overlay.test.ts | 17 ++++++++++++++++- .../src/modes/components/move-overlay.ts | 5 ++--- 3 files changed, 19 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 15ad4eaa6..b14914c17 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,7 @@ ### Fixed +- Fixed the `/move` directory picker drawing a 68-column frame inside wider full-width overlays. ([#5067](https://github.com/can1357/oh-my-pi/issues/5067)) - Fixed Go debug launches falling back to native debuggers when Delve is unavailable; nested modules and `go.work` workspaces now resolve local Delve adapters before PATH, newly installed adapters are detected without restart, and missing adapter errors include install or configuration guidance. ([#5037](https://github.com/can1357/oh-my-pi/issues/5037)) - Fixed a memory leak (large retained JavaScriptCore heaps) in the TUI during session transcript rebuilds and refreshes by properly handling snapcompact archive image frames. - Fixed a crash in interactive TUI sessions (Cannot set cwd while another same-realm JS runtime is running) when the JS evaluation worker falls back to the in-process inline path. diff --git a/packages/coding-agent/src/modes/components/__tests__/move-overlay.test.ts b/packages/coding-agent/src/modes/components/__tests__/move-overlay.test.ts index 12cfb14e6..eb5c19f1f 100644 --- a/packages/coding-agent/src/modes/components/__tests__/move-overlay.test.ts +++ b/packages/coding-agent/src/modes/components/__tests__/move-overlay.test.ts @@ -3,12 +3,14 @@ import * as fs from "node:fs"; import * as fsp from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; +import { visibleWidth } from "@oh-my-pi/pi-tui"; import { Settings } from "../../../config/settings"; import { getThemeByName, setThemeInstance, type Theme } from "../../theme/theme"; import { MoveOverlay, type MoveOverlayResult, resolveExistingDirectory, resolveMovePath } from "../move-overlay"; // Strip SGR colors so assertions see visible text only. -const strip = (lines: readonly string[]): string => lines.join("\n").replace(/\x1b\[[0-9;]*m/g, ""); +const stripAnsi = (text: string): string => text.replace(/\x1b\[[0-9;]*m/g, ""); +const strip = (lines: readonly string[]): string => lines.map(stripAnsi).join("\n"); describe("resolveMovePath", () => { it("expands ~ to homedir", () => { @@ -82,6 +84,19 @@ describe("MoveOverlay", () => { expect(text).toContain("Path:"); }); + it("renders every frame row at the assigned overlay width", () => { + const overlay = new MoveOverlay(cwd, () => {}); + const lines = overlay.render(72); + const plainLines = lines.map(stripAnsi); + + expect(lines.map(line => visibleWidth(line))).toEqual(Array(lines.length).fill(72)); + expect(plainLines[0]!.endsWith(uiTheme.boxRound.topRight)).toBe(true); + expect(plainLines.at(-1)!.endsWith(uiTheme.boxRound.bottomRight)).toBe(true); + for (const line of plainLines.slice(1, -1)) { + expect(line.endsWith(uiTheme.boxRound.vertical)).toBe(true); + } + }); + it("lists child directories (excluding hidden and files) on empty input", () => { const overlay = new MoveOverlay(cwd, () => {}); const text = strip(overlay.render(80)); diff --git a/packages/coding-agent/src/modes/components/move-overlay.ts b/packages/coding-agent/src/modes/components/move-overlay.ts index 5e35cad75..9826b97bb 100644 --- a/packages/coding-agent/src/modes/components/move-overlay.ts +++ b/packages/coding-agent/src/modes/components/move-overlay.ts @@ -25,7 +25,6 @@ interface DirEntry { } const MAX_RESULTS = 15; -const OVERLAY_WIDTH = 68; /** TTL for the directory listing cache (ms). */ const DIR_CACHE_TTL = 500; @@ -230,8 +229,8 @@ export class MoveOverlay implements Component, Focusable { } } - render(_width: number): readonly string[] { - const w = OVERLAY_WIDTH; + render(width: number): readonly string[] { + const w = width; const lines: string[] = []; lines.push(topBorder(w, "Move to directory")); From c1480b29ec3c932e880769ef7586c13d583fcee0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 13:09:34 +0000 Subject: [PATCH 077/205] fix(catalog): parsed version-first claude ids Recognized SAP AI Core hai-proxy Claude ids in claude-- order so Anthropic thinking and capability inference stay on the adaptive path. Fixes #5069 --- packages/catalog/CHANGELOG.md | 4 ++++ packages/catalog/src/identity/classify.ts | 13 ++++++---- packages/catalog/test/identity-family.test.ts | 22 +++++++++++++++++ packages/catalog/test/model-thinking.test.ts | 24 +++++++++++++++++++ 4 files changed, 59 insertions(+), 4 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index bc7d74949..2c1b83dce 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed SAP AI Core Claude ids in version-first order (`anthropic--claude-4.8-opus`) parsing as unknown, restoring Anthropic adaptive thinking metadata and capability gates. ([#5069](https://github.com/can1357/oh-my-pi/issues/5069)) + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/catalog/src/identity/classify.ts b/packages/catalog/src/identity/classify.ts index 7028b56e5..ae078ff03 100644 --- a/packages/catalog/src/identity/classify.ts +++ b/packages/catalog/src/identity/classify.ts @@ -106,15 +106,20 @@ export const parseGeminiModel = parser((modelId): GeminiModel | null => { }); export const parseAnthropicModel = parser((modelId): AnthropicModel | null => { - const match = /claude-(opus|sonnet|fable|mythos)-(\d{1,2}(?:[.-]\d{1,2}){0,2})\b/.exec(modelId); - if (!match) { + const kindFirst = /claude-(opus|sonnet|fable|mythos)-(\d{1,2}(?:[.-]\d{1,2}){0,2})\b/.exec(modelId); + const versionFirst = kindFirst + ? null + : /claude-(\d{1,2}(?:[.-]\d{1,2}){0,2})-(opus|sonnet|fable|mythos)\b/.exec(modelId); + const kind = kindFirst?.[1] ?? versionFirst?.[2]; + const versionInput = kindFirst?.[2] ?? versionFirst?.[1]; + if (!kind || !versionInput) { return null; } - const version = parseSemVer(match[2]); + const version = parseSemVer(versionInput); if (!version) { return null; } - return { family: "anthropic", kind: match[1] as AnthropicKind, version }; + return { family: "anthropic", kind: kind as AnthropicKind, version }; }); export const parseOpenAIModel = parser((modelId): OpenAIModel | null => { diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts index c0748fd8f..27dca3369 100644 --- a/packages/catalog/test/identity-family.test.ts +++ b/packages/catalog/test/identity-family.test.ts @@ -12,6 +12,7 @@ import { isOpenAIModelId, isReasoningGlmModelId, modelFamilyToken, + parseAnthropicModel, supportsAdaptiveThinkingDisplay, supportsMidConversationSystemMessages, } from "@oh-my-pi/pi-catalog/identity"; @@ -61,6 +62,22 @@ describe("isClaudeModelId", () => { }); }); +describe("parseAnthropicModel", () => { + test("parses SAP hai-proxy version-first Claude ids without accepting Haiku", () => { + expect(parseAnthropicModel("anthropic--claude-4.8-opus")).toEqual({ + family: "anthropic", + kind: "opus", + version: { major: 4, minor: 8, patch: 0 }, + }); + expect(parseAnthropicModel("anthropic--claude-4.6-opus")).toEqual({ + family: "anthropic", + kind: "opus", + version: { major: 4, minor: 6, patch: 0 }, + }); + expect(parseAnthropicModel("anthropic--claude-4.8-haiku")).toBeNull(); + }); +}); + describe("supportsAdaptiveThinkingDisplay", () => { test("allows Claude Fable 5, Opus 4.7 or newer, and Sonnet 5 or newer only", () => { expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true); @@ -71,6 +88,8 @@ describe("supportsAdaptiveThinkingDisplay", () => { // Dotted and dashed version separators are equivalent. expect(supportsAdaptiveThinkingDisplay("claude-opus-4.7")).toBe(true); expect(supportsAdaptiveThinkingDisplay("anthropic/claude-opus-4.8")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("anthropic--claude-4.8-opus")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("anthropic--claude-4.6-opus")).toBe(false); expect(supportsAdaptiveThinkingDisplay("claude-opus-4-6")).toBe(false); expect(supportsAdaptiveThinkingDisplay("claude-opus-4.6")).toBe(false); expect(supportsAdaptiveThinkingDisplay("claude-opus-4-20250514")).toBe(false); @@ -83,6 +102,7 @@ describe("hasOpus47ApiRestrictions", () => { expect(hasOpus47ApiRestrictions("claude-fable-5")).toBe(true); expect(hasOpus47ApiRestrictions("claude-opus-4-7")).toBe(true); expect(hasOpus47ApiRestrictions("claude-opus-4.8")).toBe(true); + expect(hasOpus47ApiRestrictions("anthropic--claude-4.7-opus")).toBe(true); expect(hasOpus47ApiRestrictions("claude-sonnet-5")).toBe(true); expect(hasOpus47ApiRestrictions("us.anthropic.claude-sonnet-5")).toBe(true); expect(hasOpus47ApiRestrictions("claude-opus-4-6")).toBe(false); @@ -97,6 +117,8 @@ describe("supportsMidConversationSystemMessages", () => { expect(supportsMidConversationSystemMessages("claude-opus-4-8")).toBe(true); expect(supportsMidConversationSystemMessages("claude-sonnet-5")).toBe(true); expect(supportsMidConversationSystemMessages("us.anthropic.claude-sonnet-5")).toBe(true); + expect(supportsMidConversationSystemMessages("anthropic--claude-4.8-opus")).toBe(true); + expect(supportsMidConversationSystemMessages("anthropic--claude-4.7-opus")).toBe(false); expect(supportsMidConversationSystemMessages("claude-opus-4-7")).toBe(false); expect(supportsMidConversationSystemMessages("claude-sonnet-4-6")).toBe(false); }); diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index ed08575f1..578a6afe3 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -228,6 +228,30 @@ describe("model thinking derivation", () => { expect(openRouterAnthropic.thinking?.effortMap).toBeUndefined(); }); + it("derives Anthropic adaptive thinking for SAP hai-proxy version-first Claude ids", () => { + const opus48 = createModel({ + id: "anthropic--claude-4.8-opus", + api: "anthropic-messages", + provider: "custom", + }); + const opus46 = createModel({ + id: "anthropic--claude-4.6-opus", + api: "anthropic-messages", + provider: "custom", + }); + + expect(opus48.thinking).toEqual({ + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + supportsDisplay: true, + }); + expect(getSupportedEfforts(opus48)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus46.thinking).toEqual({ + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], + }); + }); + it("maps GLM-5.2 reasoning effort per host dialect", () => { const zai = createModel({ id: "glm-5.2", From 74c63fa6c53e135fe1b2883e74e68f7bf9a1ca87 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 14:00:59 +0000 Subject: [PATCH 078/205] fix(agent): labeled system steering skips accurately - Carried steering queue origin through mid-batch interrupt polling. - Preserved queued-user skip wording while labeling advisor/system steering as system advisory skips. - Added regression coverage for advisor steering skip wording. Fixes #5074 --- packages/agent/CHANGELOG.md | 4 + packages/agent/src/agent-loop.ts | 39 +++++++-- packages/agent/src/agent.ts | 14 +++- packages/agent/src/types.ts | 17 +++- packages/agent/test/agent-loop.test.ts | 106 +++++++++++++++++++++++++ 5 files changed, 171 insertions(+), 9 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 972e4f3a4..a478b05ec 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed skipped sibling tool results caused by system advisor steering so they no longer claim a queued user message caused the skip. ([#5074](https://github.com/can1357/oh-my-pi/issues/5074)) + ## [16.4.0] - 2026-07-10 ### Added diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 7e32f52f5..f7ec38d3c 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -66,6 +66,8 @@ import type { AgentToolResult, AgentTurnEndContext, AsideMessage, + SteeringInterruptSource, + SteeringQueueState, StreamFn, } from "./types"; import { isSoftToolRequirement } from "./types"; @@ -1800,7 +1802,7 @@ async function executeToolCalls( const interruptibleSignal: AbortSignal = signal ? AbortSignal.any([signal, steeringAbortController.signal, ircAbortController.signal]) : AbortSignal.any([steeringAbortController.signal, ircAbortController.signal]); - const interruptState = { triggered: false }; + const interruptState: { triggered: boolean; source?: SteeringInterruptSource | "irc" } = { triggered: false }; const records = toolCalls.map(toolCall => { // Tools emitted via OpenAI's custom-tool path (e.g. `apply_patch` on GPT-5) @@ -1835,15 +1837,25 @@ async function executeToolCalls( // integration only provides getSteeringMessages(), the queue drains at the // injection boundary below; polling it here would strand or drop messages. let steeringQueued = false; + let steeringSource: SteeringInterruptSource | undefined; if (hasSteeringMessages) { - steeringQueued = await hasSteeringMessages(); + const queuedState = await hasSteeringMessages(); + if (typeof queuedState === "boolean") { + steeringQueued = queuedState; + steeringSource = queuedState ? "user" : undefined; + } else { + const state: SteeringQueueState = queuedState; + steeringQueued = state.queued; + steeringSource = state.source ?? (state.queued ? "unknown" : undefined); + } } if (steeringQueued) { - // User steering upgrades an in-flight IRC interrupt: it aborts the + // Queued steering upgrades an in-flight IRC interrupt: it aborts the // shared signal so foreground tools stop as they do for a user Esc. // Idempotent — a second steer poll after the abort is a no-op. if (!steeringAbortController.signal.aborted) { interruptState.triggered = true; + interruptState.source = steeringSource ?? "unknown"; steeringAbortController.abort(); } return; @@ -1855,6 +1867,7 @@ async function executeToolCalls( // Peer IRC only aborts interruptible waits: a foreground bash / write // mid-execution keeps running so we never leave partial side effects. interruptState.triggered = true; + interruptState.source = "irc"; ircAbortController.abort(); } }; @@ -2115,7 +2128,7 @@ async function executeToolCalls( // This tool's own signal fired AND it failed — it was cut off before producing // a usable result, so report it as skipped. record.skipped = true; - emitToolResult(record, createSkippedToolResult(), true); + emitToolResult(record, createSkippedToolResult(interruptState.source), true); } else { // No interrupt on this signal, or the tool finished (successfully or with a // genuine error) before the interrupt landed. Keep its real result: a completed @@ -2209,7 +2222,7 @@ async function executeToolCalls( toolName: record.toolCall.name, status: "skipped", }); - emitToolResult(record, createSkippedToolResult(), true); + emitToolResult(record, createSkippedToolResult(interruptState.source), true); } } @@ -2326,12 +2339,24 @@ function createToolSignalAbortedResult(signal: AbortSignal): AgentToolResult { +function createSkippedToolResult(source: SteeringInterruptSource | "irc" | undefined): AgentToolResult { + let reason = "pending steering message"; + let blocker = "queued message"; + if (source === "user") { + reason = "queued user message"; + blocker = "queued message"; + } else if (source === "system") { + reason = "pending system advisory"; + blocker = "advisory"; + } else if (source === "irc") { + reason = "pending peer interrupt"; + blocker = "interrupt"; + } return { content: [ { type: "text", - text: "Skipped due to queued user message. Do not count this skipped result as completed work or verification. After the queued message is handled on the next step, retry the skipped tool if it is still needed.", + text: `Skipped due to ${reason}. Do not count this skipped result as completed work or verification. After the ${blocker} is handled on the next step, retry the skipped tool if it is still needed.`, }, ], details: {}, diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 2e2b6e872..280109497 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -1184,7 +1184,19 @@ export class Agent { } return this.#dequeueSteeringMessages(); }, - hasSteeringMessages: () => this.#steeringQueue.length > 0, + hasSteeringMessages: () => { + if (this.#steeringQueue.length === 0) { + return { queued: false }; + } + for (const message of this.#steeringQueue) { + const role = "role" in message ? message.role : undefined; + const attribution = "attribution" in message ? message.attribution : undefined; + if (role === "user" && attribution !== "agent") { + return { queued: true, source: "user" }; + } + } + return { queued: true, source: "system" }; + }, hasIrcInterrupts: this.hasIrcInterrupts, getFollowUpMessages: async () => this.#dequeueFollowUpMessages(), getAsideMessages: async () => (await this.#asideMessageProvider?.()) ?? [], diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index 3a556e11b..30234b352 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -83,6 +83,17 @@ export function isSoftToolRequirement(directive: ToolChoiceDirective | undefined return typeof directive === "object" && directive !== null && (directive as SoftToolRequirement).soft === true; } +/** Source category for a queued steering interrupt observed without consuming the queue. */ +export type SteeringInterruptSource = "user" | "system" | "unknown"; + +/** Non-consuming summary of whether queued steering should interrupt a tool batch. */ +export interface SteeringQueueState { + /** True when at least one steering message is queued. */ + queued: boolean; + /** Best-effort origin used only to word synthetic skipped-tool results. */ + source?: SteeringInterruptSource; +} + /** * Configuration for the agent loop. */ @@ -194,10 +205,14 @@ export interface AgentLoopConfig extends SimpleStreamOptions { * restore queued messages while in-flight tools settle, and an external * abort in that window leaves the queue intact for a post-abort continue. * + * Returning `true` is treated as user-originated steering for compatibility. + * Return a {@link SteeringQueueState} when the queue can distinguish system + * advisories from real user messages. + * * When omitted, steering never interrupts a running tool batch; queued * messages are still delivered at the next injection boundary. */ - hasSteeringMessages?: () => boolean | Promise; + hasSteeringMessages?: () => boolean | SteeringQueueState | Promise; /** * Peeks whether IRC messages should interrupt an interruptible waiting tool. diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index 0ee5f2faa..10b65eeb9 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -21,6 +21,19 @@ import { INTENT_FIELD } from "@oh-my-pi/pi-wire"; import { type } from "arktype"; import { createAssistantMessage, createUserMessage } from "./helpers"; +declare module "@oh-my-pi/pi-agent-core/types" { + interface CustomAgentMessages { + advisor: { + role: "custom"; + customType: "advisor"; + content: string; + display: boolean; + attribution: "agent"; + timestamp: number; + }; + } +} + // Simple identity converter for tests - just passes through standard messages function identityConverter(messages: AgentMessage[]): Message[] { return messages.filter(m => m.role === "user" || m.role === "assistant" || m.role === "toolResult") as Message[]; @@ -1194,6 +1207,99 @@ describe("agentLoop with AgentMessage", () => { expect(sawInterruptInContext).toBe(true); }); + it("should skip remaining tool calls with system advisory wording when advisor steering is queued", async () => { + const toolSchema = type({ value: "string" }); + const executed: string[] = []; + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + concurrency: "exclusive", + async execute(_toolCallId, params) { + executed.push(params.value); + return { + content: [{ type: "text", text: `ok:${params.value}` }], + details: { value: params.value }, + }; + }, + }; + + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + + const advisorMessage: AgentMessage = { + role: "custom", + customType: "advisor", + content: "pause before continuing", + display: true, + attribution: "agent", + timestamp: Date.now(), + }; + let advisorDelivered = false; + + const mock = createMockModel({ + responses: [ + { + content: [ + { type: "toolCall", id: "tool-1", name: "echo", arguments: { value: "first" } }, + { type: "toolCall", id: "tool-2", name: "echo", arguments: { value: "second" } }, + ], + }, + { content: ["done"] }, + ], + }); + + const config: AgentLoopConfig = { + model: mock.model, + convertToLlm: identityConverter, + interruptMode: "immediate", + hasSteeringMessages: () => { + if (executed.length < 1 || advisorDelivered) { + return { queued: false }; + } + return { queued: true, source: "system" }; + }, + getSteeringMessages: async () => { + if (executed.length >= 1 && !advisorDelivered) { + advisorDelivered = true; + return [advisorMessage]; + } + return []; + }, + }; + + const events: AgentEvent[] = []; + const stream = agentLoop([createUserMessage("start")], context, config, undefined, mock.stream); + for await (const event of stream) { + events.push(event); + } + + expect(executed).toEqual(["first"]); + + const toolEnds = events.filter( + (e): e is Extract => e.type === "tool_execution_end", + ); + expect(toolEnds.length).toBe(2); + expect(toolEnds[0].isError).toBe(false); + expect(toolEnds[1].isError).toBe(true); + const skippedContent = toolEnds[1].result.content[0]; + expect(skippedContent?.type).toBe("text"); + if (skippedContent?.type !== "text") throw new Error("skipped tool result must be text"); + expect(skippedContent.text).toContain("Skipped due to pending system advisory"); + expect(skippedContent.text).not.toContain("queued user message"); + expect(skippedContent.text).toContain("Do not count this skipped result as completed work"); + expect(skippedContent.text).toContain("retry the skipped tool if it is still needed"); + + const advisorInjected = events.some( + event => + event.type === "message_start" && + event.message.role === "custom" && + event.message.customType === "advisor" && + event.message.content === "pause before continuing", + ); + expect(advisorInjected).toBe(true); + }); + it("drains queued steering by aborting an interruptible tool mid-wait", async () => { const toolSchema = type({}); let steerReady = false; From a0a6949a4a6381dabe9bc2e6146c531184c6aa27 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 15:04:19 +0000 Subject: [PATCH 079/205] fix(mcp): matched stdio spawn overload for tcc Switched stdio MCP server launches to the argv-first Bun.spawn overload used by JS eval so macOS TCC Apple Events prompts reach children like xcrun mcpbridge. Added regression coverage for the spawn call shape. Fixes #5085 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/mcp/transports/stdio.test.ts | 46 ++++++++++++++++++- .../coding-agent/src/mcp/transports/stdio.ts | 9 ++-- 3 files changed, 55 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 04ae4eb4d..0a805b9bd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed macOS stdio MCP servers still missing Apple Events TCC prompts by spawning MCP children through the same argv-first `Bun.spawn` overload used by the JS eval kernel. ([#5085](https://github.com/can1357/oh-my-pi/issues/5085)) + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/coding-agent/src/mcp/transports/stdio.test.ts b/packages/coding-agent/src/mcp/transports/stdio.test.ts index ac1364787..17bbd2d6d 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.test.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from "bun:test"; +import { describe, expect, it, spyOn } from "bun:test"; import { resolveStdioSpawnCommand, StdioTransport } from "./stdio"; @@ -56,6 +56,50 @@ describe("resolveStdioSpawnCommand", () => { }); }); +describe("StdioTransport.connect", () => { + it("passes argv as Bun.spawn's first argument and process options as the second", async () => { + const cwd = process.cwd(); + const envValue = "stdio-spawn-shape"; + const argv = [process.execPath, "-e", "process.exit(0)"]; + const transport = new StdioTransport({ + command: argv[0], + args: argv.slice(1), + cwd, + env: { + OMP_STDIO_SPAWN_SHAPE: envValue, + }, + }); + const spawnSpy = spyOn(Bun, "spawn"); + + try { + await transport.connect(); + + expect(spawnSpy).toHaveBeenCalledTimes(1); + const call = spawnSpy.mock.calls[0]; + if (!call) throw new Error("expected StdioTransport.connect() to spawn exactly one subprocess"); + + const [spawnArgv, spawnOptions] = call; + expect(spawnArgv).toEqual(argv); + expect(spawnOptions).toEqual( + expect.objectContaining({ + cwd, + detached: !(process.platform === "darwin" || process.platform === "win32"), + env: expect.objectContaining({ + OMP_STDIO_SPAWN_SHAPE: envValue, + }), + stderr: "pipe", + stdin: "pipe", + stdout: "pipe", + windowsHide: process.platform === "win32" ? expect.any(Boolean) : undefined, + }), + ); + } finally { + await transport.close(); + spawnSpy.mockRestore(); + } + }); +}); + // Regression for #3945: request() awaited stdin.write/flush, so a child that // stops draining stdin would park the async fn past the timeout timer and past // `return promise`, orphaning the deferred rejection and hanging the caller diff --git a/packages/coding-agent/src/mcp/transports/stdio.ts b/packages/coding-agent/src/mcp/transports/stdio.ts index 228226008..dbed6ec4d 100644 --- a/packages/coding-agent/src/mcp/transports/stdio.ts +++ b/packages/coding-agent/src/mcp/transports/stdio.ts @@ -8,7 +8,7 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; import { getProjectDir, readJsonl, Snowflake } from "@oh-my-pi/pi-utils"; -import { type Subprocess, spawn } from "bun"; +import type { Subprocess } from "bun"; import { hostHasInheritableConsole } from "../../eval/py/spawn-options"; import type { JsonRpcError, @@ -376,8 +376,11 @@ export class StdioTransport implements MCPTransport { // macOS stays attached so TCC can prompt for Apple Events automation; // Windows stays attached, and only hides the child when the host has no // console to share. See `StdioSpawnCommand`. - this.#process = spawn({ - cmd: spawnCommand.cmd, + // Keep this on Bun's argv-first overload. The eval JS kernel path that + // triggers macOS Apple Events TCC prompts uses the same shape; the + // one-object `{ cmd }` overload timed out before prompting for `mcpbridge` + // even with `detached: false` (#5085). + this.#process = Bun.spawn(spawnCommand.cmd, { cwd, env, stdin: "pipe", From cf4e510acdfd0b8b874c2c93988744abd26a70d9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 15:14:06 +0000 Subject: [PATCH 080/205] fix(tool): resolved bare skill urls to directories Bare skill:// URLs now resolve to the skill directory for path-only tool operations while read still returns SKILL.md instructions. Fixes #5087 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../src/internal-urls/skill-protocol.ts | 2 +- .../coding-agent/src/tools/bash-skill-urls.ts | 2 +- packages/coding-agent/src/tools/glob.ts | 1 + .../test/tools/bash-skill-urls.test.ts | 6 +++--- .../test/tools/grep-internal-urls.test.ts | 18 ++++++++++++++++++ 6 files changed, 28 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 04ae4eb4d..a9c07b497 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed bare `skill://` path-only resolution for bash, grep, and glob so tools receive the skill directory instead of `SKILL.md`, while `read skill://` still opens the skill instructions. ([#5087](https://github.com/can1357/oh-my-pi/issues/5087)) + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/coding-agent/src/internal-urls/skill-protocol.ts b/packages/coding-agent/src/internal-urls/skill-protocol.ts index 6975c94ca..7842a87b7 100644 --- a/packages/coding-agent/src/internal-urls/skill-protocol.ts +++ b/packages/coding-agent/src/internal-urls/skill-protocol.ts @@ -72,7 +72,7 @@ export class SkillProtocolHandler implements ProtocolHandler { throw new Error("Path traversal is not allowed"); } } else { - targetPath = skill.filePath; + targetPath = context?.pathOnly === true ? skill.baseDir : skill.filePath; } let stats: fsTypes.Stats; diff --git a/packages/coding-agent/src/tools/bash-skill-urls.ts b/packages/coding-agent/src/tools/bash-skill-urls.ts index 585882a9c..6db86b693 100644 --- a/packages/coding-agent/src/tools/bash-skill-urls.ts +++ b/packages/coding-agent/src/tools/bash-skill-urls.ts @@ -66,7 +66,7 @@ export function resolveSkillUrlToPath(url: string, skills: readonly Skill[]): st const hasRelativePath = rawPath !== "" && rawPath !== "/"; if (!hasRelativePath) { - return path.resolve(skill.filePath); + return path.resolve(skill.baseDir); } let relativePath: string; diff --git a/packages/coding-agent/src/tools/glob.ts b/packages/coding-agent/src/tools/glob.ts index 79176c459..81a066598 100644 --- a/packages/coding-agent/src/tools/glob.ts +++ b/packages/coding-agent/src/tools/glob.ts @@ -184,6 +184,7 @@ export class GlobTool implements AgentTool { signal, localProtocolOptions: this.session.localProtocolOptions, skills: this.session.skills, + pathOnly: true, }); if (!resource.sourcePath) { throw new ToolError(`Cannot find internal URL without a backing file: ${rawPattern}`); diff --git a/packages/coding-agent/test/tools/bash-skill-urls.test.ts b/packages/coding-agent/test/tools/bash-skill-urls.test.ts index 153eef880..7bbf12e55 100644 --- a/packages/coding-agent/test/tools/bash-skill-urls.test.ts +++ b/packages/coding-agent/test/tools/bash-skill-urls.test.ts @@ -133,11 +133,11 @@ describe("expandSkillUrls", () => { expect(expandSkillUrls(command, skills)).toBe(`python ${shellEscape(expectedPath)}`); }); - it("resolves skill://name with no relative path to SKILL.md", () => { + it("resolves skill://name with no relative path to the skill directory", () => { const skills = [createSkill("valid-skill", "/tmp/skills/valid-skill")]; - const command = "cat skill://valid-skill"; + const command = "printf '%s\n' skill://valid-skill"; - expect(expandSkillUrls(command, skills)).toBe(`cat ${shellEscape(skills[0].filePath)}`); + expect(expandSkillUrls(command, skills)).toBe(`printf '%s\n' ${shellEscape(skills[0].baseDir)}`); }); it("returns command unchanged when no skills are loaded", () => { diff --git a/packages/coding-agent/test/tools/grep-internal-urls.test.ts b/packages/coding-agent/test/tools/grep-internal-urls.test.ts index b0d2e48ae..ec9c122df 100644 --- a/packages/coding-agent/test/tools/grep-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/grep-internal-urls.test.ts @@ -182,6 +182,24 @@ describe("GrepTool internal URL resolution", () => { expect(getResultText(findResult)).toContain("guide.md"); }); + it("walks bare skill:// roots for search and find", async () => { + await registerSkillDirectory(); + const session = createSession({ hasEditTool: true }); + const searchTool = new GrepTool(session); + const findTool = new GlobTool(session); + + const searchResult = await searchTool.execute("test-search", { + pattern: "deep needle", + path: "skill://demo", + }); + const findResult = await findTool.execute("test-find", { + path: "skill://demo", + }); + + expect(getResultText(searchResult)).toContain("deep needle"); + expect(getResultText(findResult)).toContain("guide.md"); + }); + it("resolves artifact:// URL to backing file and greps it", async () => { const content = "line one\nfound the needle here\nline three\n"; await Bun.write(path.join(artifactsDir, "5.bash.log"), content); From bce6aa89a450e2a3dd2a7bd735c7dc951fb14ea3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 18:08:28 +0000 Subject: [PATCH 081/205] fix(mcp): included OAuth scopes in dynamic client registration Clerk and similar providers bind DCR clients to only the scopes declared at registration. Authorize then requests scopes_supported (including openid), which rejects with "client is not allowed to request scope 'openid'". Match Claude Code by sending config.scopes as RFC 7591 scope on the DCR body. --- packages/coding-agent/CHANGELOG.md | 4 +++ packages/coding-agent/src/mcp/oauth-flow.ts | 25 ++++++++++++------ packages/coding-agent/test/oauth-flow.test.ts | 26 +++++++++++++++++++ 3 files changed, 47 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 04ae4eb4d..412d37450 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed MCP OAuth dynamic client registration omitting discovered scopes on the RFC 7591 registration body. Providers such as Clerk bind DCR-created clients to only the scopes declared at registration, then reject the subsequent authorize request when it asks for `openid` (from `scopes_supported`). Registration now includes `config.scopes` when present, matching Claude Code and the scopes already sent on authorize. + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/coding-agent/src/mcp/oauth-flow.ts b/packages/coding-agent/src/mcp/oauth-flow.ts index 36e6f9bf7..a456ec41c 100644 --- a/packages/coding-agent/src/mcp/oauth-flow.ts +++ b/packages/coding-agent/src/mcp/oauth-flow.ts @@ -555,12 +555,28 @@ export class MCPOAuthFlow extends OAuthCallbackFlow { * "Only clients listed in the Figma MCP Catalog can connect"), the fallback * probe surfaces a message that names the endpoint and status instead of * the historical opaque "OAuth provider requires client_id". + * + * Includes {@link MCPOAuthConfig.scopes} as RFC 7591 `scope` when set so + * providers that bind DCR clients to registered scopes only (e.g. Clerk) + * accept the later authorize request for the same scope set. */ async #tryRegisterClient(redirectUri: string): Promise { const registrationEndpoint = await this.#resolveRegistrationEndpoint(); if (!registrationEndpoint) return; try { + const registrationBody: Record = { + client_name: "oh-my-pi", + redirect_uris: [redirectUri], + grant_types: ["authorization_code", "refresh_token"], + response_types: ["code"], + token_endpoint_auth_method: "none", + application_type: "native", + }; + const scope = this.config.scopes?.trim(); + if (scope) { + registrationBody.scope = scope; + } const response = await this.#fetch(registrationEndpoint, { method: "POST", headers: { @@ -568,14 +584,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow { Accept: "application/json", }, signal: this.ctrl.signal, - body: JSON.stringify({ - client_name: "oh-my-pi", - redirect_uris: [redirectUri], - grant_types: ["authorization_code", "refresh_token"], - response_types: ["code"], - token_endpoint_auth_method: "none", - application_type: "native", - }), + body: JSON.stringify(registrationBody), }); if (!response.ok) { diff --git a/packages/coding-agent/test/oauth-flow.test.ts b/packages/coding-agent/test/oauth-flow.test.ts index d022e3ad4..ade93f5b0 100644 --- a/packages/coding-agent/test/oauth-flow.test.ts +++ b/packages/coding-agent/test/oauth-flow.test.ts @@ -80,10 +80,36 @@ describe("mcp oauth flow", () => { expect(registrationPayload).not.toBeNull(); expect((registrationPayload as { client_name?: string } | null)?.client_name).toBe("oh-my-pi"); + expect((registrationPayload as { scope?: string } | null)?.scope).toBeUndefined(); expect(authUrl.searchParams.get("client_id")).toBe("registered-client-id"); expect(authUrl.searchParams.get("state")).toBe("test-state"); }); + it("includes discovered scopes in dynamic client registration", async () => { + let registrationPayload: Record | null = null; + const scopes = "openid profile email offline_access"; + + const flow = new MCPOAuthFlow( + { + authorizationUrl: "https://www.figma.com/oauth/mcp", + tokenUrl: "https://api.figma.com/v1/oauth/token", + scopes, + fetch: mockFigmaRegistration(payload => { + registrationPayload = payload; + }), + }, + {}, + ); + + const { url } = await flow.generateAuthUrl("test-state", "http://127.0.0.1:53173/callback"); + const authUrl = new URL(url); + + expect(registrationPayload).not.toBeNull(); + expect((registrationPayload as { scope?: string } | null)?.scope).toBe(scopes); + expect(authUrl.searchParams.get("scope")).toBe(scopes); + expect(authUrl.searchParams.get("client_id")).toBe("registered-client-id"); + }); + it("omits prompt by default so provider-specific reauth pages can use returning grants", async () => { const flow = new MCPOAuthFlow( { From 060179729e59951c19e4c04c01322a2c54e6d518 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 18:08:40 +0000 Subject: [PATCH 082/205] fix(codex): sent version header with requests Added Codex version headers to turn requests and model discovery so backend-gated models resolve consistently. Fixes #5105 --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/providers/openai-codex-responses.ts | 1 + packages/ai/test/openai-codex-responses-lite.test.ts | 2 ++ packages/catalog/CHANGELOG.md | 4 ++++ packages/catalog/src/discovery/codex.ts | 5 +++-- packages/catalog/src/wire/codex.ts | 1 + packages/catalog/test/codex-discovery.test.ts | 10 +++++++--- 7 files changed, 19 insertions(+), 5 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index eda640539..327de2002 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -17,6 +17,7 @@ ### Fixed +- Fixed OpenAI Codex turn requests to include the Codex `version` header, matching upstream Codex request metadata for newly gated models. - Fixed xAI SuperGrok multi-account rotation to correctly treat HTTP 403 credit exhaustion and spending limit errors as usage limits, triggering a credential rotation to a sibling account. - Fixed error classification for AWS credential-resolution failures (AwsCredentialsError) to correctly map them as authentication failures. - Fixed OpenAI-compatible chat-completions streams to preserve vLLM-style trailing cached-token usage chunks, ensuring accurate cacheRead and billable input session statistics. diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 0fa0f720c..391cad23f 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -3774,6 +3774,7 @@ function createCodexHeaders( headers.delete("openai-beta"); headers.set(OPENAI_HEADERS.BETA, betaHeader); headers.set(OPENAI_HEADERS.ORIGINATOR, OPENAI_HEADER_VALUES.ORIGINATOR_CODEX); + headers.set(OPENAI_HEADERS.VERSION, packageJson.version); headers.set("User-Agent", `pi/${packageJson.version} (${os.platform()} ${os.release()}; ${os.arch()})`); if (sessionId) { headers.set(OPENAI_HEADERS.CONVERSATION_ID, sessionId); diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 992ee26d3..17eab4c90 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -14,6 +14,7 @@ import { isOpenAIResponsesProgressEvent } from "@oh-my-pi/pi-ai/providers/openai import type { CodexCompactionRequestContext, Context, FetchImpl, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import * as piUtils from "@oh-my-pi/pi-utils"; +import packageJson from "../package.json" with { type: "json" }; import { createCodexModel } from "./helpers"; const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; @@ -625,6 +626,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(result.stopReason).toBe("stop"); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.headers.get("version")).toBe(packageJson.version); expect(captured?.body.instructions).toBeUndefined(); expect(captured?.body.tools).toBeUndefined(); expect((captured?.body.input as Array>)[0]?.type).toBe("additional_tools"); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index bc7d74949..81eb8b5ce 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI Codex model discovery to include the Codex `version` header alongside the `client_version` query parameter. + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 5784d7088..2a06d526e 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -86,12 +86,12 @@ export async function fetchCodexModels(options: CodexModelDiscoveryOptions): Pro const fetchFn = discoveryFetch(options.fetchFn); const baseUrl = normalizeBaseUrl(options.baseUrl); const paths = normalizePaths(options.paths); - const headers = buildCodexHeaders(options); const clientVersion = await resolveCodexClientVersion( options.clientVersion, options.registryFetchFn ?? fetchFn, options.signal, ); + const headers = buildCodexHeaders(options, clientVersion); let sawSuccessfulResponse = false; for (const path of paths) { @@ -156,7 +156,7 @@ function buildModelsUrl(baseUrl: string, path: string, clientVersion: string | u return url.toString(); } -function buildCodexHeaders(options: CodexModelDiscoveryOptions): Headers { +function buildCodexHeaders(options: CodexModelDiscoveryOptions, clientVersion: string): Headers { const headers = new Headers(options.headers); headers.set("Authorization", `Bearer ${options.accessToken}`); if (options.accountId && options.accountId.trim().length > 0) { @@ -164,6 +164,7 @@ function buildCodexHeaders(options: CodexModelDiscoveryOptions): Headers { } headers.set(OPENAI_HEADERS.BETA, OPENAI_HEADER_VALUES.BETA_RESPONSES); headers.set(OPENAI_HEADERS.ORIGINATOR, OPENAI_HEADER_VALUES.ORIGINATOR_CODEX); + headers.set(OPENAI_HEADERS.VERSION, clientVersion); headers.set("accept", "application/json"); return headers; } diff --git a/packages/catalog/src/wire/codex.ts b/packages/catalog/src/wire/codex.ts index 329ace70f..48edeb392 100644 --- a/packages/catalog/src/wire/codex.ts +++ b/packages/catalog/src/wire/codex.ts @@ -8,6 +8,7 @@ export const OPENAI_HEADERS = { BETA: "OpenAI-Beta", ACCOUNT_ID: "chatgpt-account-id", ORIGINATOR: "originator", + VERSION: "version", SESSION_ID: "session_id", CONVERSATION_ID: "conversation_id", SCOPED_SESSION_ID: "session-id", diff --git a/packages/catalog/test/codex-discovery.test.ts b/packages/catalog/test/codex-discovery.test.ts index babff9c28..b647ac895 100644 --- a/packages/catalog/test/codex-discovery.test.ts +++ b/packages/catalog/test/codex-discovery.test.ts @@ -11,9 +11,11 @@ import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; describe("Codex model discovery", () => { it("marks discovered models for provider-native V2 compaction", async () => { + let capturedHeaders: Headers | undefined; const fetchFn: typeof fetch = Object.assign( - async () => - new Response( + async (_input: string | URL | Request, init?: RequestInit) => { + capturedHeaders = new Headers(init?.headers); + return new Response( JSON.stringify({ models: [ { @@ -28,7 +30,8 @@ describe("Codex model discovery", () => { ], }), { headers: { etag: "models-v1" } }, - ), + ); + }, { preconnect() {} }, ); const result = await fetchCodexModels({ @@ -38,6 +41,7 @@ describe("Codex model discovery", () => { fetchFn, }); + expect(capturedHeaders?.get("version")).toBe("0.99.0"); expect(result?.etag).toBe("models-v1"); expect(result?.models).toHaveLength(1); expect(result?.models[0]).toMatchObject({ From c30bdc54ce2a1bd1d975fc2815e0206d1e570d18 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 20:09:31 +0200 Subject: [PATCH 083/205] feat(ai): enforced all_turns reasoning context for responses lite - Force `reasoning.context` to `all_turns` in OpenAI Codex requests when `responsesLite` is enabled. - Include `reasoning.encrypted_content` in the `include` header for lite responses. - Update request transformation logic to ensure these fields are populated even when reasoning effort is not explicitly set. --- packages/agent/CHANGELOG.md | 4 ++ .../src/compaction/compaction-v2-streaming.ts | 6 ++- packages/agent/src/compaction/openai.ts | 10 +++++ packages/agent/test/remote-compaction.test.ts | 8 +++- packages/ai/CHANGELOG.md | 4 ++ .../openai-codex/request-transformer.ts | 10 +++-- .../test/openai-codex-responses-lite.test.ts | 44 ++++++++++++++++++- 7 files changed, 77 insertions(+), 9 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 972e4f3a4..e0989db65 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Enabled reasoning encryption content for all Responses Lite compaction requests + ## [16.4.0] - 2026-07-10 ### Added diff --git a/packages/agent/src/compaction/compaction-v2-streaming.ts b/packages/agent/src/compaction/compaction-v2-streaming.ts index 14a2318f8..8514d4717 100644 --- a/packages/agent/src/compaction/compaction-v2-streaming.ts +++ b/packages/agent/src/compaction/compaction-v2-streaming.ts @@ -305,10 +305,12 @@ async function attemptCompactionV2Streaming( instructions: request.instructions, stream: true, store: false, - ...(request.reasoning + ...(request.reasoning || model.useResponsesLite ? { // Lite implies gpt-5.4+, where codex-rs sends `all_turns` replay. - reasoning: model.useResponsesLite ? { ...request.reasoning, context: "all_turns" } : request.reasoning, + reasoning: model.useResponsesLite + ? { ...(request.reasoning ?? {}), context: "all_turns" } + : request.reasoning, include: ["reasoning.encrypted_content"], } : {}), diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index fbac0a8b7..aeba4603c 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -88,6 +88,11 @@ export interface OpenAiRemoteCompactionRequest { model: string; input: Array>; instructions: string; + reasoning?: { + context?: string; + [key: string]: unknown; + }; + include?: string[]; } export interface OpenAiRemoteCompactionResponse extends OpenAiRemoteCompactionPreserveData {} @@ -533,6 +538,11 @@ export async function requestOpenAiRemoteCompaction( if (model.useResponsesLite) { applyCodexResponsesLiteShape(request); headers[OPENAI_HEADERS.RESPONSES_LITE] = "true"; + request.reasoning = { + ...request.reasoning, + context: "all_turns", + }; + request.include = Array.from(new Set([...(request.include ?? []), "reasoning.encrypted_content"])); } } diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index 78d4cb97b..6051b467b 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -446,6 +446,8 @@ describe("Responses Lite remote compaction", () => { tools?: unknown; input?: Array>; client_metadata?: unknown; + reasoning?: Record; + include?: string[]; } interface CapturedLiteExchange { @@ -501,8 +503,9 @@ describe("Responses Lite remote compaction", () => { ); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.reasoning).toEqual({ context: "all_turns" }); + expect(captured?.body.include).toEqual(["reasoning.encrypted_content"]); expect(captured?.body.instructions).toBeUndefined(); - expect(captured?.body.tools).toBeUndefined(); expect(captured?.body.client_metadata).toBeUndefined(); expect(captured?.headers.get("x-codex-installation-id")).toBe(TEST_INSTALLATION_ID); expect(captured?.headers.get("session-id")).toBe("codex-compaction-session"); @@ -552,8 +555,9 @@ describe("Responses Lite remote compaction", () => { }); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.reasoning).toEqual({ context: "all_turns" }); + expect(captured?.body.include).toEqual(["reasoning.encrypted_content"]); expect(captured?.body.instructions).toBeUndefined(); - expect(captured?.body.tools).toBeUndefined(); if (!isRecord(captured?.body.client_metadata)) throw new Error("expected V2 client_metadata"); const v2ClientMetadata = captured.body.client_metadata; const v2TurnMetadata = parseCodexTurnMetadata(v2ClientMetadata["x-codex-turn-metadata"]); diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index eda640539..8b65af82f 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- Enforced `all_turns` reasoning context for all Responses Lite requests + ## [16.4.0] - 2026-07-10 ### Added diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index a25d5b665..9131cde81 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -32,7 +32,7 @@ export interface CodexRequestOptions { /** User-facing effort; maps 1:1 onto the wire tier of the same name. */ reasoningEffort?: CodexCallerEffort | "none"; reasoningSummary?: ReasoningConfig["summary"] | null; - /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ + /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. Gated to gpt-5.4+ Codex models (older ids reject it, so it is suppressed and `context` omitted). Note that under Responses Lite (`responsesLite`), the server strictly requires `reasoning.context` to be `all_turns`, which overrides this option and forces `all_turns`. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; @@ -379,8 +379,9 @@ export async function transformRequestBody( applyCodexResponsesLiteShape(body); } - if (options.reasoningEffort !== undefined) { - const reasoningConfig = getReasoningConfig(model, options.reasoningEffort, options); + if (options.reasoningEffort !== undefined || responsesLite) { + const reasoningConfig = + options.reasoningEffort !== undefined ? getReasoningConfig(model, options.reasoningEffort, options) : {}; body.reasoning = { ...body.reasoning, ...reasoningConfig, @@ -394,7 +395,8 @@ export async function transformRequestBody( // default. The version gate is authoritative: even an explicit // `all_turns` override is suppressed on unsupported models, while // `current_turn`/`auto` (universally supported) always pass through. - const context = options.reasoningContext ?? "all_turns"; + // Note: Responses Lite forces `all_turns` to satisfy the transport's server invariant. + const context = responsesLite ? "all_turns" : (options.reasoningContext ?? "all_turns"); if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) { delete body.reasoning.context; } else { diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 992ee26d3..2a56e4a59 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -142,7 +142,47 @@ describe("openai-codex reasoning.context", () => { responsesLite: true, reasoningContext: "auto", }); - expect(overridden.reasoning?.context).toBe("auto"); + expect(overridden.reasoning?.context).toBe("all_turns"); + }); + + it("enforces reasoning.context to be all_turns for the lite transport even when effort is unset or none", async () => { + const model = createCodexModel("gpt-5.5"); + + // Case 1: reasoningEffort is undefined (missing effort) + const missingEffort = await transformRequestBody({ model: model.id }, model, { + responsesLite: true, + }); + expect(missingEffort.reasoning?.context).toBe("all_turns"); + expect(missingEffort.reasoning?.effort).toBeUndefined(); + + // Case 2: reasoningEffort is explicitly "none" (effort set to off) + const noneEffort = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "none", + responsesLite: true, + }); + expect(noneEffort.reasoning?.context).toBe("all_turns"); + expect(noneEffort.reasoning?.effort).toBe("none"); + + // Case 3: Conflicting explicit reasoningContext with missing effort under Lite + const conflictingUnsetEffort = await transformRequestBody({ model: model.id }, model, { + responsesLite: true, + reasoningContext: "current_turn", + }); + expect(conflictingUnsetEffort.reasoning?.context).toBe("all_turns"); + + // Case 4: Conflicting explicit reasoningContext with "none" effort under Lite + const conflictingNoneEffort = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "none", + responsesLite: true, + reasoningContext: "current_turn", + }); + expect(conflictingNoneEffort.reasoning?.context).toBe("all_turns"); + + // Case 5: responsesLite is false and reasoningEffort is undefined (regular request with no effort) + const plainRequest = await transformRequestBody({ model: model.id }, model, { + responsesLite: false, + }); + expect(plainRequest.reasoning).toBeUndefined(); }); // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns` @@ -599,6 +639,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(result.stopReason).toBe("stop"); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.reasoning).toEqual({ context: "all_turns" }); expect(captured?.body.input).toEqual([ { type: "additional_tools", role: "developer", tools: [] }, { @@ -625,6 +666,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(result.stopReason).toBe("stop"); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.reasoning).toEqual({ context: "all_turns" }); expect(captured?.body.instructions).toBeUndefined(); expect(captured?.body.tools).toBeUndefined(); expect((captured?.body.input as Array>)[0]?.type).toBe("additional_tools"); From d9854ade76e2db640949f75b35bfffa46eab7eb6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 20:21:05 +0200 Subject: [PATCH 084/205] feat(catalog): updated model definitions and provider constraints - Added `gpt-5.6-luna`, `gpt-5.6-sol`, and `gpt-5.6-terra` variants for the `opencode-zen` provider. - Removed deprecated `*-pro` model aliases from the `openai-codex` provider in `models.json`. - Updated `openai-compat.ts` to restrict pro-reasoning alias generation to the `openai` provider only. - Adjusted various model `contextWindow`, `maxTokens`, and `cost` parameters to reflect latest upstream metadata. - Added `perplexity-academic-researcher` model definition. --- packages/catalog/CHANGELOG.md | 16 + packages/catalog/src/models.json | 366 +++++++++++------- .../src/provider-models/openai-compat.ts | 28 +- 3 files changed, 260 insertions(+), 150 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index bc7d74949..90858532b 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,22 @@ ## [Unreleased] +### Added + +- Added GPT-5.6 Luna, Sol, and Terra models +- Added perplexity-academic-researcher model + +### Changed + +- Updated context windows for multiple GPT-5.6 models +- Increased max tokens for several models +- Updated cache write costs for GPT-5.6 variants +- Reduced pricing for select models + +### Removed + +- Removed pro-reasoning aliases for GPT-5.6 variants + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 7eecd2bd0..80bb13350 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -18667,7 +18667,7 @@ "cacheRead": 0.2, "cacheWrite": 0 }, - "contextWindow": 200000, + "contextWindow": 1000000, "maxTokens": 64000, "headers": { "User-Agent": "opencode/1.3.15", @@ -19190,7 +19190,7 @@ "cacheRead": 0.5, "cacheWrite": 0 }, - "contextWindow": 400000, + "contextWindow": 1050000, "maxTokens": 128000, "headers": { "User-Agent": "opencode/1.3.15", @@ -19207,6 +19207,108 @@ }, "contextPromotionTarget": "github-copilot/gpt-5.4" }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-responses", + "provider": "github-copilot", + "baseUrl": "https://api.githubcopilot.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "headers": { + "User-Agent": "opencode/1.3.15", + "X-GitHub-Api-Version": "2026-06-01" + }, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-responses", + "provider": "github-copilot", + "baseUrl": "https://api.githubcopilot.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "headers": { + "User-Agent": "opencode/1.3.15", + "X-GitHub-Api-Version": "2026-06-01" + }, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-responses", + "provider": "github-copilot", + "baseUrl": "https://api.githubcopilot.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "headers": { + "User-Agent": "opencode/1.3.15", + "X-GitHub-Api-Version": "2026-06-01" + }, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, "grok-code-fast-1": { "id": "grok-code-fast-1", "name": "Grok Code Fast 1", @@ -48660,6 +48762,25 @@ "contextWindow": null, "maxTokens": null }, + "perplexity-academic-researcher": { + "id": "perplexity-academic-researcher", + "name": "perplexity-academic-researcher", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "phi-4-mini-instruct": { "id": "phi-4-mini-instruct", "name": "phi-4-mini-instruct", @@ -63390,47 +63511,6 @@ ] } }, - "gpt-5.6-luna-pro": { - "id": "gpt-5.6-luna-pro", - "name": "GPT-5.6 Luna Pro", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.1, - "cacheWrite": 1.25 - }, - "remoteCompaction": { - "enabled": true, - "api": "openai-codex-responses", - "v2StreamingEnabled": true - }, - "contextWindow": 372000, - "maxTokens": 128000, - "preferWebsockets": true, - "useResponsesLite": true, - "priority": 3, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro", - "applyPatchToolType": "freeform", - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - } - }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", @@ -63470,47 +63550,6 @@ ] } }, - "gpt-5.6-sol-pro": { - "id": "gpt-5.6-sol-pro", - "name": "GPT-5.6 Sol Pro", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 5, - "output": 30, - "cacheRead": 0.5, - "cacheWrite": 6.25 - }, - "remoteCompaction": { - "enabled": true, - "api": "openai-codex-responses", - "v2StreamingEnabled": true - }, - "contextWindow": 372000, - "maxTokens": 128000, - "preferWebsockets": true, - "useResponsesLite": true, - "priority": 1, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro", - "applyPatchToolType": "freeform", - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - } - }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", @@ -63549,47 +63588,6 @@ "max" ] } - }, - "gpt-5.6-terra-pro": { - "id": "gpt-5.6-terra-pro", - "name": "GPT-5.6 Terra Pro", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.5, - "output": 15, - "cacheRead": 0.25, - "cacheWrite": 3.125 - }, - "remoteCompaction": { - "enabled": true, - "api": "openai-codex-responses", - "v2StreamingEnabled": true - }, - "contextWindow": 372000, - "maxTokens": 128000, - "preferWebsockets": true, - "useResponsesLite": true, - "priority": 2, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro", - "applyPatchToolType": "freeform", - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - } } }, "opencode": { @@ -65525,6 +65523,96 @@ }, "contextPromotionTarget": "opencode-zen/gpt-5.4" }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-responses", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-responses", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-responses", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, "grok-4.5": { "id": "grok-4.5", "name": "Grok 4.5", @@ -65601,7 +65689,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 256000, + "contextWindow": 190000, "maxTokens": 64000, "thinking": { "mode": "effort", @@ -68111,13 +68199,13 @@ "text" ], "cost": { - "input": 0.09, - "output": 0.18, - "cacheRead": 0.018, + "input": 0.08399999999999999, + "output": 0.16799999999999998, + "cacheRead": 0.016800000000000002, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 65536, + "maxTokens": 384000, "thinking": { "mode": "effort", "efforts": [ @@ -75529,9 +75617,9 @@ "text" ], "cost": { - "input": 0.77, - "output": 2.42, - "cacheRead": 0.143, + "input": 0.42, + "output": 1.32, + "cacheRead": 0.078, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -79347,7 +79435,7 @@ "input": 1.25, "output": 7.5, "cacheRead": 0.125, - "cacheWrite": 0 + "cacheWrite": 1.5625 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -79377,7 +79465,7 @@ "input": 1.25, "output": 7.5, "cacheRead": 0.125, - "cacheWrite": 0 + "cacheWrite": 1.5625 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -79407,7 +79495,7 @@ "input": 6.25, "output": 37.5, "cacheRead": 0.625, - "cacheWrite": 0 + "cacheWrite": 7.8125 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -79437,7 +79525,7 @@ "input": 6.25, "output": 37.5, "cacheRead": 0.625, - "cacheWrite": 0 + "cacheWrite": 7.8125 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -79467,7 +79555,7 @@ "input": 3.125, "output": 18.75, "cacheRead": 0.3125, - "cacheWrite": 0 + "cacheWrite": 3.90625 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -79497,7 +79585,7 @@ "input": 3.125, "output": 18.75, "cacheRead": 0.3125, - "cacheWrite": 0 + "cacheWrite": 3.90625 }, "contextWindow": 1000000, "maxTokens": 128000, diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 8952fcb78..ef7f7eafa 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -823,17 +823,23 @@ const OPENAI_PRO_REASONING_BASE_IDS: Record = { "gpt-5.6-sol": true, "gpt-5.6-terra": true, }; -const OPENAI_PRO_REASONING_PROVIDERS: Record = { openai: true, "openai-codex": true }; +/** + * Providers whose generated pro aliases this pass owns. `openai-codex` stays in + * the sweep so stale aliases from earlier snapshots are dropped on regen, but + * projection is `openai`-only — subscription (Codex) auth does not offer pro + * reasoning. + */ +const OPENAI_PRO_REASONING_SWEEP_PROVIDERS: Record = { openai: true, "openai-codex": true }; /** * A row this generator pass owns: one of the derived `gpt-5.6-*-pro` alias ids - * on `openai`/`openai-codex` that carries the generated `reasoningMode` marker. + * on a swept provider that carries the generated `reasoningMode` marker. * A real upstream model occupying the same id has no `reasoningMode` and is * never touched. */ function isGeneratedOpenAIProReasoningAlias(model: ModelSpec): boolean { return ( - OPENAI_PRO_REASONING_PROVIDERS[model.provider] === true && + OPENAI_PRO_REASONING_SWEEP_PROVIDERS[model.provider] === true && model.reasoningMode !== undefined && model.id.endsWith("-pro") && OPENAI_PRO_REASONING_BASE_IDS[model.id.slice(0, -"-pro".length)] === true @@ -842,21 +848,21 @@ function isGeneratedOpenAIProReasoningAlias(model: ModelSpec): boolean { /** * Re-derive the generated pro-reasoning aliases (`gpt-5.6-*-pro`) for the - * first-party `openai`/`openai-codex` gpt-5.6 rows. Each alias inherits the - * base row's metadata, requests the base wire id via `requestModelId`, and - * sets `reasoningMode: "pro"` so Responses-family request builders emit + * first-party `openai` gpt-5.6 rows. Each alias inherits the base row's + * metadata, requests the base wire id via `requestModelId`, and sets + * `reasoningMode: "pro"` so Responses-family request builders emit * `reasoning: { mode: "pro" }`. Called by the models.json generator after all - * sources merge: stale copies of the owned aliases (previous snapshot) are - * dropped and re-projected from the current base rows so alias metadata always - * tracks the base, while a real upstream model that occupies an alias id wins - * and suppresses the projection. + * sources merge: stale copies of the owned aliases (previous snapshot, + * including retired `openai-codex` rows) are dropped and re-projected from the + * current base rows so alias metadata always tracks the base, while a real + * upstream model that occupies an alias id wins and suppresses the projection. */ export function projectOpenAIProReasoningAliases(models: readonly ModelSpec[]): ModelSpec[] { const kept = models.filter(model => !isGeneratedOpenAIProReasoningAlias(model)); const ids = new Set(kept.map(model => `${model.provider}/${model.id}`)); const out = [...kept]; for (const model of kept) { - if (!OPENAI_PRO_REASONING_PROVIDERS[model.provider]) continue; + if (model.provider !== "openai") continue; if (!OPENAI_PRO_REASONING_BASE_IDS[model.id]) continue; const aliasId = `${model.id}-pro`; const aliasKey = `${model.provider}/${aliasId}`; From 1624d48dd82b9198862b373ecfc4f79aefa85102 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 18:23:06 +0000 Subject: [PATCH 085/205] fix(codex): used codex version for turns Resolved the Codex client version for turn requests with the same catalog resolver used by discovery. Fixes #5105 --- .../ai/src/providers/openai-codex-responses.ts | 14 +++++++++++++- .../ai/test/openai-codex-responses-lite.test.ts | 6 ++++-- packages/catalog/src/discovery/codex.ts | 2 +- 3 files changed, 18 insertions(+), 4 deletions(-) diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 391cad23f..67b578d3c 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -1,5 +1,6 @@ import * as os from "node:os"; import { scheduler } from "node:timers/promises"; +import { resolveCodexClientVersion } from "@oh-my-pi/pi-catalog/discovery/codex"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { CODEX_BASE_URL, @@ -632,6 +633,7 @@ interface CodexRequestContext { baseUrl: string; url: string; requestHeaders: Record; + codexClientVersion: string; transportSessionId?: string; providerSessionState?: CodexProviderSessionState; isolatedTransportState?: CodexProviderSessionState; @@ -1200,6 +1202,7 @@ async function buildCodexRequestContext( const url = resolveCodexResponsesUrl(baseUrl); const promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); + const codexClientVersion = await resolveCodexClientVersion(undefined, options?.fetch ?? fetch, options?.signal); const transformedBody = await buildTransformedCodexRequestBody(model, context, options, promptCacheKey); const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) }; @@ -1275,6 +1278,7 @@ async function buildCodexRequestContext( websocketState, responsesLite, requestMetadata, + codexClientVersion, transformedBody, rawRequestDump, }; @@ -1427,6 +1431,7 @@ async function openCodexWebSocketTransport( requestContext.requestHeaders, requestContext.accountId, requestContext.apiKey, + requestContext.codexClientVersion, requestContext.transportSessionId, "websocket", websocketState, @@ -1527,6 +1532,7 @@ async function openCodexSseTransport( wireBody, state, requestContext.responsesLite, + requestContext.codexClientVersion, requestContext.requestMetadata, requestSetup.requestSignal, requestSetup.firstEventTimeoutMs, @@ -2496,6 +2502,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses" baseUrl: model.baseUrl || CODEX_BASE_URL, url: "", requestHeaders: {}, + codexClientVersion: packageJson.version, responsesLite: options?.responsesLite === true, transformedBody: { model: model.id }, rawRequestDump: { @@ -2559,6 +2566,7 @@ export async function prewarmOpenAICodexResponses( transportSessionId ?? crypto.randomUUID(), providerSessionState, ); + const codexClientVersion = await resolveCodexClientVersion(undefined, fetch, options?.signal); const requestIdentity = createCodexCompatibilityIdentity(metadataSession); const headers = logger.time( "prewarmCodex:createHeaders", @@ -2566,6 +2574,7 @@ export async function prewarmOpenAICodexResponses( { ...(model.headers ?? {}), ...(options?.headers ?? {}) }, accountId, apiKey, + codexClientVersion, promptCacheKey, "websocket", state, @@ -3685,6 +3694,7 @@ async function openCodexSseEventStream( body: RequestBody, state: CodexWebSocketSessionState | undefined, responsesLite: boolean, + codexClientVersion: string, requestMetadata: CodexRequestMetadata | undefined, signal: AbortSignal | undefined, firstEventTimeoutMs: number | undefined, @@ -3695,6 +3705,7 @@ async function openCodexSseEventStream( requestHeaders, accountId, apiKey, + codexClientVersion, sessionId, "sse", state, @@ -3756,6 +3767,7 @@ function createCodexHeaders( initHeaders: Record | undefined, accountId: string | undefined, accessToken: string, + codexClientVersion: string, sessionId?: string, transport: CodexTransport = "sse", state?: CodexWebSocketSessionState, @@ -3774,7 +3786,7 @@ function createCodexHeaders( headers.delete("openai-beta"); headers.set(OPENAI_HEADERS.BETA, betaHeader); headers.set(OPENAI_HEADERS.ORIGINATOR, OPENAI_HEADER_VALUES.ORIGINATOR_CODEX); - headers.set(OPENAI_HEADERS.VERSION, packageJson.version); + headers.set(OPENAI_HEADERS.VERSION, codexClientVersion); headers.set("User-Agent", `pi/${packageJson.version} (${os.platform()} ${os.release()}; ${os.arch()})`); if (sessionId) { headers.set(OPENAI_HEADERS.CONVERSATION_ID, sessionId); diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 17eab4c90..3b77febe4 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -14,7 +14,6 @@ import { isOpenAIResponsesProgressEvent } from "@oh-my-pi/pi-ai/providers/openai import type { CodexCompactionRequestContext, Context, FetchImpl, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import * as piUtils from "@oh-my-pi/pi-utils"; -import packageJson from "../package.json" with { type: "json" }; import { createCodexModel } from "./helpers"; const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; @@ -101,6 +100,9 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe if (url === "https://api.github.com/repos/openai/codex/releases/latest") { return new Response(JSON.stringify({ tag_name: "rust-v0.0.0" }), { status: 200 }); } + if (url === "https://registry.npmjs.org/@openai%2Fcodex/latest") { + return new Response(JSON.stringify({ version: "0.144.1" }), { status: 200 }); + } if (url.startsWith("https://raw.githubusercontent.com/openai/codex/")) { return new Response("PROMPT", { status: 200, headers: { etag: '"etag"' } }); } @@ -626,7 +628,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(result.stopReason).toBe("stop"); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); - expect(captured?.headers.get("version")).toBe(packageJson.version); + expect(captured?.headers.get("version")).toBe("0.144.1"); expect(captured?.body.instructions).toBeUndefined(); expect(captured?.body.tools).toBeUndefined(); expect((captured?.body.input as Array>)[0]?.type).toBe("additional_tools"); diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 2a06d526e..118e37fde 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -169,7 +169,7 @@ function buildCodexHeaders(options: CodexModelDiscoveryOptions, clientVersion: s return headers; } -async function resolveCodexClientVersion( +export async function resolveCodexClientVersion( clientVersion: string | undefined, fetchFn: FetchImpl, signal: AbortSignal | undefined, From 60c9ad625ec063c3fbc032111485b77b0eeeed32 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 20:32:05 +0200 Subject: [PATCH 086/205] feat(coding-agent/prompts): updated advisor system prompt behavior - Instruct the advisor to stop policing scope, ambition, or backwards compatibility unless explicitly requested by the user. - Adjust the blocker criteria to require explicit contradictions of user instructions rather than subjective assessments of refactor size or scope. --- packages/catalog/CHANGELOG.md | 2 +- packages/coding-agent/CHANGELOG.md | 5 +++++ packages/coding-agent/src/prompts/advisor/system.md | 10 +++++++++- 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 90858532b..d7efd673c 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -16,7 +16,7 @@ ### Removed -- Removed pro-reasoning aliases for GPT-5.6 variants +- Removed the generated GPT-5.6 pro-reasoning aliases (`gpt-5.6-{luna,sol,terra}-pro`) from the `openai-codex` subscription provider — pro reasoning is not offered on subscriptions; the `openai` API-key aliases remain ## [16.4.0] - 2026-07-10 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 412d37450..6e6c767c2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Changed + +- Reduced agent bias against large diffs and refactors in advisor prompts +- Updated advisor blocker criteria to prioritize explicit user instructions over plan size + ### Fixed - Fixed MCP OAuth dynamic client registration omitting discovered scopes on the RFC 7591 registration body. Providers such as Clerk bind DCR-created clients to only the scopes declared at registration, then reject the subsequent authorize request when it asks for `openid` (from `scopes_supported`). Registration now includes `config.scopes` when present, matching Claude Code and the scopes already sent on authorize. diff --git a/packages/coding-agent/src/prompts/advisor/system.md b/packages/coding-agent/src/prompts/advisor/system.md index 981aada05..0d930193b 100644 --- a/packages/coding-agent/src/prompts/advisor/system.md +++ b/packages/coding-agent/src/prompts/advisor/system.md @@ -44,6 +44,14 @@ NEVER advise on intent or process: - Intent is the agent's domain; it defaults to informed action. - Your lane: correctness, edge cases, design, process. +NEVER police scope or ambition: +- A large diff, wholesale rewrite, or expanding plan is NOT a problem by itself — often it is exactly what the user wants. +- Object to the size or reach of a change ONLY when it contradicts an explicit user instruction in the transcript (e.g. "minimal change", "don't touch X") — and cite that instruction. + +NEVER raise backwards compatibility unless the user or a standing project rule explicitly requires it: +- No unsolicited concerns or blockers about breaking changes, deprecation shims, migration paths, legacy fallbacks, or API stability. +- Absent such a requirement, clean cutover — delete the old path, update every caller — is the correct default; treat it as such. + Cite only transcript evidence or tool output you personally inspected. Arguments absent from the rendered transcript are UNKNOWN: - NEVER assert concrete values, array indexes, serialization shapes, or caller mistakes for hidden arguments. @@ -76,7 +84,7 @@ Cite the exact instruction or risk. **`blocker`** - Stop and reconsider. - Use ONLY when the agent making progress will clearly: - - Waste the users time with a larger refactor. + - Contradict an explicit user instruction in the transcript — cite it; size, rewrite breadth, or an evolving plan alone is NEVER the trigger. - Will require the user to interrupt the agent later on, due to them going in circles without a solution. - Be fundamentally unsound. - Hand off as "done" work that was never exercised against the user's actual ask. From 7cbe9fc0a88965a870719655b2f8e68ef196df6a Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 20:40:04 +0200 Subject: [PATCH 087/205] feat(tui): implemented grid-based rendering engine for complex math layouts - Introduced a grid-based rendering engine to support complex LaTeX environments like matrices, cases, and arrays. - Added advanced layout capabilities including stretched delimiters, multi-line radicals, and limits-based stacking for operators. - Implemented logic to propagate color scopes and font styles across box boundaries. - Exported math font commands and structural glyph painting to facilitate consistent rendering in 2-D contexts. --- packages/tui/CHANGELOG.md | 13 + packages/tui/src/latex-block.ts | 901 ++++++++++++++++++++++-- packages/tui/src/latex-to-unicode.ts | 23 + packages/tui/test/latex-block.test.ts | 184 ++++- packages/tui/test/markdown-math.test.ts | 27 +- packages/tui/test/markdown.test.ts | 6 +- 6 files changed, 1065 insertions(+), 89 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index e794cb8c5..7e6b2eb9e 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,19 @@ ## [Unreleased] +### Added + +- Added full 2-D layout support for display LaTeX math (fractions, matrices, radicals, limits), modeled on the layout approach of [txm](https://github.com/thatmagicalcat/txm) (Terminal TeX Math) by [@thatmagicalcat](https://github.com/thatmagicalcat) +- Added support for `\left`, `\right`, and `\middle` stretchy delimiters in display blocks +- Added rendering for `cases`, `matrix`, `pmatrix`, `bmatrix`, and `vmatrix` environments +- Added support for block-level scripts and big-operator limits (e.g., `\sum`, `\int\limits`) +- Added cross-box styling for `\color`, `\textcolor`, and math font commands (e.g., `\mathbf`) + +### Changed + +- Improved row alignment and spacing for `align`, `gather`, and `array` environments +- Updated matrix environments to render as baseline-aligned grids with stretched brackets + ## [16.4.0] - 2026-07-10 ### Fixed diff --git a/packages/tui/src/latex-block.ts b/packages/tui/src/latex-block.ts index 43477fec7..32a9ce190 100644 --- a/packages/tui/src/latex-block.ts +++ b/packages/tui/src/latex-block.ts @@ -1,16 +1,27 @@ -// Two-dimensional layout for *display* LaTeX math: stacks `\frac` numerator over -// denominator with a horizontal bar, aligning surrounding text to the bar's row. +// Two-dimensional layout engine for *display* LaTeX math. // -// −b ± √(b² − 4ac) -// x = ──────────────── -// 2a +// ┌───────── n ⎛ a+b ⎞² +// −b ± ╲│ b² − 4ac ∑ xᵢ ⎜ ───── ⎟ ⎡ 1 2 ⎤ +// x = ────────────────── i=0 ⎝ c ⎠ ⎣ 3 4 ⎦ +// 2a // // Only display blocks (`$$…$$`, `\[…\]`) use this; inline `$…$` stays single-line -// (`½`, `(a+b)/c`). Everything that is not a fraction — symbols, scripts, roots, -// matrices, environments — is delegated to `latexToUnicode`, so this engine only -// adds the vertical stacking the flat string form can't express. +// via `latexToUnicode` (`½`, `(a+b)/c`). The engine lays out a `Box` tree — +// rectangles of padded lines with a `baseline` row — and knows how to stack +// fractions and `\binom`, stretch delimiters (`\left…\right`, tall bare parens, +// matrix brackets), render matrix/cases/array environments as baseline-aligned +// grids, place big-operator limits (`\sum`, `\lim`, `\int\limits`) above and +// below the symbol, draw radicals, raise/lower block scripts, and +// align `&` columns in `align`-family environments. Flat runs — symbols, fonts, +// colors, inline scripts — are delegated to `latexToUnicode`. +// +// The 2-D layout approach (stretchy delimiter piecing, stacked operator limits, +// baseline-aligned matrix grids, drawn radicals, block scripts) is modeled on +// txm — Terminal TeX Math — by @thatmagicalcat +// (https://github.com/thatmagicalcat/txm, MIT/Apache-2.0), reimplemented from +// scratch here on this module's ANSI-aware Box model. -import { latexToUnicode } from "./latex-to-unicode"; +import { latexColorScope, latexToUnicode, MATH_FONT_COMMANDS } from "./latex-to-unicode"; import { visibleWidth } from "./utils"; /** @@ -24,13 +35,15 @@ interface Box { width: number; } +type CellAlign = "l" | "c" | "r"; + const BAR = "─"; const FRAC_COMMANDS: Record = { frac: true, dfrac: true, tfrac: true, cfrac: true }; +const BINOM_COMMANDS: Record = { binom: true, dbinom: true, tbinom: true }; // Display "wrapper" environments whose body is an expression (possibly with `\\` -// row breaks and `&` alignment). Their bodies are parsed so fractions inside -// stack; grid/structure environments (matrix/array/cases) stay opaque and are -// rendered flat by `latexToUnicode`. +// row breaks and `&` alignment). Their rows are parsed so fractions inside stack +// and `&` columns align. const DISPLAY_ROW_ENVIRONMENTS: Record = { equation: true, eqnarray: true, @@ -48,6 +61,140 @@ const DISPLAY_ROW_ENVIRONMENTS: Record = { math: true, }; +// Environments laid out as 2-D grids of parsed cells: [open, close] delimiter. +const GRID_ENVIRONMENTS: Record = { + matrix: ["", ""], + smallmatrix: ["", ""], + array: ["", ""], + pmatrix: ["(", ")"], + bmatrix: ["[", "]"], + Bmatrix: ["{", "}"], + vmatrix: ["|", "|"], + Vmatrix: ["‖", "‖"], + cases: ["{", ""], + dcases: ["{", ""], + rcases: ["", "}"], + drcases: ["", "}"], +}; + +// Operators whose display-style scripts stack above/below the symbol. +const LIMIT_OPERATORS: Record = { + sum: true, + prod: true, + coprod: true, + bigcup: true, + bigcap: true, + bigsqcup: true, + bigvee: true, + bigwedge: true, + bigoplus: true, + bigotimes: true, + bigodot: true, + biguplus: true, + lim: true, + limsup: true, + liminf: true, + projlim: true, + injlim: true, + varlimsup: true, + varliminf: true, + varprojlim: true, + varinjlim: true, + max: true, + min: true, + sup: true, + inf: true, + det: true, + gcd: true, + Pr: true, + argmax: true, + argmin: true, +}; + +// Integral-family operators: scripts stay beside the symbol (LaTeX display +// convention) unless an explicit `\limits` follows. +const INTEGRAL_OPERATORS: Record = { + int: true, + iint: true, + iiint: true, + iiiint: true, + oint: true, + oiint: true, + oiiint: true, + idotsint: true, + intop: true, + smallint: true, +}; + +// Vertical delimiter piece characters: `only` for single-line content, then +// top/mid/bot columns for stretched forms; `axis` replaces `mid` at the +// baseline row (the brace point). +interface DelimPieces { + only: string; + top: string; + mid: string; + bot: string; + axis?: string; +} + +const DELIM_PIECES: Record = { + "(": { only: "(", top: "⎛", mid: "⎜", bot: "⎝" }, + ")": { only: ")", top: "⎞", mid: "⎟", bot: "⎠" }, + "[": { only: "[", top: "⎡", mid: "⎢", bot: "⎣" }, + "]": { only: "]", top: "⎤", mid: "⎥", bot: "⎦" }, + "{": { only: "{", top: "⎧", mid: "⎪", bot: "⎩", axis: "⎨" }, + "}": { only: "}", top: "⎫", mid: "⎪", bot: "⎭", axis: "⎬" }, + "|": { only: "|", top: "│", mid: "│", bot: "│" }, + "‖": { only: "‖", top: "║", mid: "║", bot: "║" }, + "⌈": { only: "⌈", top: "⎡", mid: "⎢", bot: "⎢" }, + "⌉": { only: "⌉", top: "⎤", mid: "⎥", bot: "⎥" }, + "⌊": { only: "⌊", top: "⎢", mid: "⎢", bot: "⎣" }, + "⌋": { only: "⌋", top: "⎥", mid: "⎥", bot: "⎦" }, +}; + +// `\left`/`\right`/`\middle` delimiter token → piece-table key. Unknown tokens +// fall back to `latexToUnicode` and render at the baseline row only. +const DELIM_KEYS: Record = { + "(": "(", + ")": ")", + "[": "[", + "]": "]", + "\\{": "{", + "\\}": "}", + "\\lbrace": "{", + "\\rbrace": "}", + "|": "|", + "\\vert": "|", + "\\lvert": "|", + "\\rvert": "|", + "\\|": "‖", + "\\Vert": "‖", + "\\lVert": "‖", + "\\rVert": "‖", + "\\langle": "⟨", + "\\rangle": "⟩", + "<": "⟨", + ">": "⟩", + "\\lceil": "⌈", + "\\rceil": "⌉", + "\\lfloor": "⌊", + "\\rfloor": "⌋", + "\\lbrack": "[", + "\\rbrack": "]", + ".": "", +}; + +/** + * Inline-run conversion context. `wrap` re-applies the scoped commands (math + * fonts, colors) active at this point in the parse, so each flat run handed to + * `latexToUnicode` renders with the same styling it would have had in one piece. + */ +interface Ctx { + wrap: (run: string) => string; +} + +const ROOT_CTX: Ctx = { wrap: run => run }; + function spaces(n: number): string { return n > 0 ? " ".repeat(n) : ""; } @@ -73,6 +220,19 @@ function textBox(text: string): Box { return { lines: raw.map(line => padRight(line, width)), baseline: (raw.length - 1) >> 1, width }; } +/** Pad every line of `b` to `width` per `align`, keeping the baseline. */ +function padBox(b: Box, width: number, align: CellAlign): Box { + if (b.width >= width) return b; + const lines = b.lines.map(line => { + const extra = width - visibleWidth(line); + if (align === "l") return line + spaces(extra); + if (align === "r") return spaces(extra) + line; + const left = extra >> 1; + return spaces(left) + line + spaces(extra - left); + }); + return { lines, baseline: b.baseline, width }; +} + /** Place boxes side by side, aligning their baselines. */ function hconcat(boxes: Box[]): Box { if (boxes.length === 1) return boxes[0]; @@ -97,6 +257,18 @@ function hconcat(boxes: Box[]): Box { return { lines, baseline: above, width }; } +/** Stack boxes vertically, e.g. the rows of an aligned block. */ +function vconcat(boxes: Box[], align: CellAlign = "l"): Box { + if (boxes.length === 1) return boxes[0]; + let width = 0; + for (const b of boxes) width = Math.max(width, b.width); + const lines: string[] = []; + for (const b of boxes) { + for (const line of b.lines) lines.push(align === "c" ? center(line, width) : padRight(line, width)); + } + return { lines, baseline: (lines.length - 1) >> 1, width }; +} + /** Stack `num` over `den`, separated by a bar; the bar becomes the baseline. */ function fracBox(num: Box, den: Box): Box { const width = Math.max(num.width, den.width) + 2; @@ -108,14 +280,157 @@ function fracBox(num: Box, den: Box): Box { return { lines, baseline: num.lines.length, width }; } -/** Stack boxes vertically (left-aligned), e.g. the rows of an aligned block. */ -function vconcat(boxes: Box[]): Box { - if (boxes.length === 1) return boxes[0]; - let width = 0; - for (const b of boxes) width = Math.max(width, b.width); +/** + * One vertical delimiter column of `height` rows for piece-table key `key` + * (`"("`, `"{"`, …); null when `key` is empty (`\left.`). Unknown keys render a + * single glyph at the baseline row. + */ +function delimColumn(key: string, height: number, baseline: number): Box | null { + if (!key) return null; + const pieces = DELIM_PIECES[key]; + if (height <= 1) { + const only = pieces?.only ?? key; + return only ? { lines: [only], baseline: 0, width: visibleWidth(only) } : null; + } + const width = visibleWidth(pieces?.only ?? key); + const blank = spaces(width); const lines: string[] = []; - for (const b of boxes) for (const line of b.lines) lines.push(padRight(line, width)); - return { lines, baseline: (lines.length - 1) >> 1, width }; + if (!pieces) { + for (let y = 0; y < height; y++) lines.push(y === baseline ? key : blank); + return { lines, baseline, width }; + } + const axisRow = Math.min(Math.max(baseline, 1), height - 2); + for (let y = 0; y < height; y++) { + if (y === 0) lines.push(pieces.top); + else if (y === height - 1) lines.push(pieces.bot); + else if (y === axisRow && pieces.axis) lines.push(pieces.axis); + else lines.push(pieces.mid); + } + return { lines, baseline, width }; +} + +/** Wrap `inner` in (possibly stretched) delimiters, padding tall content. */ +function delimBox(inner: Box, left: string, right: string): Box { + const height = inner.lines.length; + const lcol = delimColumn(left, height, inner.baseline); + const rcol = delimColumn(right, height, inner.baseline); + if (!lcol && !rcol) return inner; + const pad: Box | null = height > 1 ? textBox(" ") : null; + const parts: Box[] = []; + if (lcol) parts.push(lcol); + if (pad) parts.push(pad); + parts.push(inner); + if (pad) parts.push(pad); + if (rcol) parts.push(rcol); + return hconcat(parts); +} + +/** `\binom{n}{k}`: `n` over `k` (no bar) inside stretched parentheses. */ +function binomBox(top: Box, bottom: Box): Box { + const width = Math.max(top.width, bottom.width); + const lines = [ + ...top.lines.map(line => center(line, width)), + spaces(width), + ...bottom.lines.map(line => center(line, width)), + ]; + return delimBox({ lines, baseline: top.lines.length, width }, "(", ")"); +} + +/** + * A drawn radical for a multi-line radicand: overline row on top, bar column + * on the left, hook at the bottom. Single-line radicands stay flat (`√x̄`). + */ +function radicalBox(inner: Box, degree: string | null): Box { + const lines: string[] = [` ┌${BAR.repeat(inner.width + 1)}`]; + for (let y = 0; y < inner.lines.length; y++) { + lines.push((y === inner.lines.length - 1 ? "╲│ " : " │ ") + inner.lines[y]); + } + const box: Box = { lines, baseline: inner.baseline + 1, width: inner.width + 3 }; + if (!degree) return box; + const deg = latexToUnicode(`^{${degree}}`); + // Degree sits one row above the baseline, at the radical's upper left. + return hconcat([{ lines: [deg, spaces(visibleWidth(deg))], baseline: 1, width: visibleWidth(deg) }, box]); +} + +/** Big operator with limits: `sup` centered above `glyph`, `sub` below. */ +function limitsBox(glyph: Box, sub: Box | null, sup: Box | null): Box { + const width = Math.max(glyph.width, sub?.width ?? 0, sup?.width ?? 0); + const lines: string[] = []; + if (sup) for (const line of sup.lines) lines.push(center(line, width)); + const baseline = lines.length + glyph.baseline; + for (const line of glyph.lines) lines.push(center(line, width)); + if (sub) for (const line of sub.lines) lines.push(center(line, width)); + return { lines, baseline, width }; +} + +/** + * Attach block scripts to `base` as one shared right-hand column: the + * superscript ends level with the base's top row (raised one row above a + * single-line base), the subscript starts level with its bottom row (lowered + * one row below a single-line base). + */ +function attachScripts(base: Box, sub: Box | null, sup: Box | null): Box { + if (sub === null && sup === null) return base; + const single = base.lines.length === 1; + const width = Math.max(sub?.width ?? 0, sup?.width ?? 0); + const blank = spaces(width); + const lines: string[] = []; + let baseline = 0; + if (sup) { + const lift = single ? 1 : base.baseline; + for (const line of sup.lines) lines.push(padRight(line, width)); + for (let k = 0; k < lift; k++) lines.push(blank); + baseline = lines.length - 1; + } + if (sub) { + const below = base.lines.length - 1 - base.baseline - (sub.lines.length - 1); + let drop = Math.max(below, single ? 1 : 0); + if (sup && drop < 1) drop = 1; + // Rows between the baseline row and the subscript's top row. + const gap = lines.length === 0 ? drop : drop - 1; + for (let k = 0; k < gap; k++) lines.push(blank); + for (const line of sub.lines) lines.push(padRight(line, width)); + } + return hconcat([base, { lines, baseline, width }]); +} + +/** + * Lay out parsed cells as a grid: per-column width/alignment, per-gap width. + * With `rowGap > 0` (matrix-family environments), blank rows separate the grid + * rows and the total height is forced odd, so the baseline sits at the true + * vertical center — `A = [matrix]` centers on the brackets, and stretched + * braces get a real middle piece even for two content rows. + */ +function gridBox(rows: Box[][], align: (col: number) => CellAlign, gap: (col: number) => number, rowGap = 0): Box { + let ncols = 0; + for (const row of rows) ncols = Math.max(ncols, row.length); + if (ncols === 0 || rows.length === 0) return textBox(""); + const widths = new Array(ncols).fill(0); + for (const row of rows) { + row.forEach((cell, j) => { + widths[j] = Math.max(widths[j], cell.width); + }); + } + const rowBoxes: Box[] = []; + for (const row of rows) { + if (rowGap > 0 && rowBoxes.length > 0) { + for (let g = 0; g < rowGap; g++) rowBoxes.push({ lines: [""], baseline: 0, width: 0 }); + } + const parts: Box[] = []; + for (let j = 0; j < ncols; j++) { + if (j > 0) { + const g = gap(j); + if (g > 0) parts.push({ lines: [spaces(g)], baseline: 0, width: g }); + } + parts.push(padBox(row[j] ?? { lines: [""], baseline: 0, width: 0 }, widths[j], align(j))); + } + rowBoxes.push(hconcat(parts)); + } + const grid = vconcat(rowBoxes); + if (rowGap > 0 && rows.length > 1 && grid.lines.length % 2 === 0) { + return { lines: [...grid.lines, spaces(grid.width)], baseline: grid.lines.length >> 1, width: grid.width }; + } + return grid; } interface Span { @@ -155,7 +470,7 @@ function readBraceGroup(src: string, i: number): Span { } /** - * Read one fraction argument: a `{…}` group, a single char, or a `\command` + * Read one command argument: a `{…}` group, a single char, or a `\command` * together with its attached `[…]`/`{…}` arguments (or whole `\begin…\end` * block), so e.g. `\frac\sqrt{a}{b}` reads `\sqrt{a}` as the numerator. */ @@ -186,6 +501,100 @@ function readArg(src: string, i: number): Span { return { text: src.slice(i, end), end }; } +/** Read a `\left`/`\right`/`\middle` delimiter token (char or `\command`). */ +function readDelimToken(src: string, i: number): Span | null { + while (src[i] === " ") i++; + if (i >= src.length) return null; + if (src[i] !== "\\") return { text: src[i], end: i + 1 }; + let j = i + 1; + if (!/[A-Za-z]/.test(src[j] ?? "")) return { text: src.slice(i, j + 1), end: j + 1 }; + while (/[A-Za-z]/.test(src[j] ?? "")) j++; + return { text: src.slice(i, j), end: j }; +} + +/** Piece-table key for a delimiter token; unknown commands resolve via Unicode. */ +function delimKey(token: string): string { + const mapped = DELIM_KEYS[token]; + if (mapped !== undefined) return mapped; + return token.startsWith("\\") ? latexToUnicode(token).trim() : token; +} + +interface LeftRightParts { + left: string; + /** Inner source split at top-level `\middle` delimiters. */ + segments: string[]; + middles: string[]; + right: string; + end: number; +} + +/** Parse `\left⟨tok⟩ … \right⟨tok⟩` starting at the backslash of `\left`. */ +function readLeftRight(src: string, start: number): LeftRightParts | null { + const left = readDelimToken(src, start + 5); + if (!left) return null; + const segments: string[] = []; + const middles: string[] = []; + let depth = 1; + let k = left.end; + let segStart = k; + while (k < src.length) { + if (src[k] !== "\\") { + k++; + continue; + } + if (src.startsWith("\\left", k) && !/[A-Za-z]/.test(src[k + 5] ?? "")) { + depth++; + const tok = readDelimToken(src, k + 5); + k = tok ? tok.end : k + 5; + continue; + } + if (src.startsWith("\\right", k) && !/[A-Za-z]/.test(src[k + 6] ?? "")) { + depth--; + const tok = readDelimToken(src, k + 6); + if (depth === 0) { + segments.push(src.slice(segStart, k)); + return { left: left.text, segments, middles, right: tok ? tok.text : ".", end: tok ? tok.end : k + 6 }; + } + k = tok ? tok.end : k + 6; + continue; + } + if (depth === 1 && src.startsWith("\\middle", k) && !/[A-Za-z]/.test(src[k + 7] ?? "")) { + segments.push(src.slice(segStart, k)); + const tok = readDelimToken(src, k + 7); + middles.push(tok ? tok.text : "|"); + k = segStart = tok ? tok.end : k + 7; + continue; + } + k += 2; // escaped char / other command head — never a boundary + } + return null; // unbalanced +} + +/** + * Index of the `close` matching the `open` at `i`, skipping escapes and brace + * groups; −1 when unbalanced (e.g. interval notation `[0, 1)`). + */ +function matchDelim(src: string, i: number, open: string, close: string): number { + let depth = 0; + for (let k = i; k < src.length; k++) { + const c = src[k]; + if (c === "\\") { + k++; + continue; + } + if (c === "{") { + k = readBraceGroup(src, k).end - 1; + continue; + } + if (c === open) depth++; + else if (c === close) { + depth--; + if (depth === 0) return k; + } + } + return -1; +} + interface EnvParts { env: string; bodyStart: number; @@ -270,31 +679,39 @@ function splitRows(body: string): string[] { return rows; } -/** - * Render a `\begin{env}…\end{env}` block. Expression "wrapper" environments - * (`equation`, `align`, `gather`, …) have their rows parsed so fractions stack; - * grid/structure environments (matrix/array/cases) render flat via - * `latexToUnicode`. - */ -function parseEnvironment(src: string, start: number): { box: Box; end: number } | null { - const env = readEnvironment(src, start); - if (env === null) return null; - const base = env.env.endsWith("*") ? env.env.slice(0, -1) : env.env; - if (!DISPLAY_ROW_ENVIRONMENTS[base]) { - return { box: textBox(latexToUnicode(src.slice(start, env.end))), end: env.end }; +/** Split a row on top-level `&` column separators (depth-aware), trimming cells. */ +function splitCells(row: string): string[] { + const cells: string[] = []; + let braceDepth = 0; + let envDepth = 0; + let last = 0; + let i = 0; + while (i < row.length) { + if (row.startsWith("\\begin", i)) { + envDepth++; + i += 6; + continue; + } + if (row.startsWith("\\end", i)) { + envDepth--; + i += 4; + continue; + } + const c = row[i]; + if (c === "\\") { + i += 2; // `\&` and command heads never split + continue; + } + if (c === "{") braceDepth++; + else if (c === "}") braceDepth--; + else if (c === "&" && braceDepth === 0 && envDepth === 0) { + cells.push(row.slice(last, i)); + last = i + 1; + } + i++; } - let bodyStart = env.bodyStart; - if (base === "alignat" || base === "alignedat" || base === "gatheredat") { - // These carry a required column-count argument `{n}` before the body. - let p = bodyStart; - while (src[p] === " " || src[p] === "\n") p++; - if (src[p] === "{") bodyStart = readBraceGroup(src, p).end; - } - const rows = splitRows(src.slice(bodyStart, env.bodyEnd)) - .map(row => row.trim()) - .filter(row => row !== "") - .map(row => parseExpr(row)); - return { box: rows.length > 0 ? vconcat(rows) : textBox(""), end: env.end }; + cells.push(row.slice(last)); + return cells.map(cell => cell.trim()); } /** Append a script (`^`/`_`) and its argument to the inline run verbatim. */ @@ -319,21 +736,122 @@ function readScript(src: string, i: number): Span { return { text: out, end: i }; } +/** Bare argument of a script read by `readScript` (`^{ab}` → `ab`, `^a` → `a`). */ +function scriptArgOf(text: string): string { + let arg = text.slice(1).trimStart(); + if (arg.startsWith("{") && arg.endsWith("}")) arg = arg.slice(1, -1); + return arg; +} + /** - * Parse a math fragment into a layout box, stacking top-level fractions (and - * fractions nested inside other fractions' arguments). Non-fraction runs — - * including scripts, roots, environments, and command arguments — are gathered - * into inline strings and rendered through `latexToUnicode`. + * Render a `\begin{env}…\end{env}` block. Grid environments (matrix family, + * cases, array) become baseline-aligned 2-D grids in stretched delimiters; + * wrapper environments (`align`, `gather`, …) parse each `\\` row, aligning `&` + * columns; anything else (tabular, …) renders flat via `latexToUnicode`. */ -function parseExpr(src: string): Box { +function parseEnvironment(src: string, start: number, ctx: Ctx): { box: Box; end: number } | null { + const env = readEnvironment(src, start); + if (env === null) return null; + const starred = env.env.endsWith("*"); + const base = starred ? env.env.slice(0, -1) : env.env; + const gridDelims = GRID_ENVIRONMENTS[base]; + if (gridDelims) { + let p = env.bodyStart; + while (src[p] === " " || src[p] === "\n" || src[p] === "\t") p++; + if (starred && src[p] === "[") { + // Starred matrix variants take an optional alignment argument. + const close = src.indexOf("]", p); + if (close !== -1 && close < env.bodyEnd) { + p = close + 1; + while (src[p] === " " || src[p] === "\n" || src[p] === "\t") p++; + } + } + let colSpec: CellAlign[] | null = null; + if (base === "array" && src[p] === "{") { + const spec = readBraceGroup(src, p); + colSpec = [...spec.text].filter((ch): ch is CellAlign => ch === "l" || ch === "c" || ch === "r"); + p = spec.end; + } + const cells = splitRows(src.slice(p, env.bodyEnd)) + .map(row => row.trim()) + .filter(row => row !== "") + .map(row => splitCells(row).map(cell => parseExpr(cell, ctx))); + const isCases = base === "cases" || base === "dcases" || base === "rcases" || base === "drcases"; + const align: (col: number) => CellAlign = colSpec ? col => colSpec[col] ?? "c" : isCases ? () => "l" : () => "c"; + const grid = gridBox(cells, align, () => 2, 1); + return { box: delimBox(grid, gridDelims[0], gridDelims[1]), end: env.end }; + } + if (!DISPLAY_ROW_ENVIRONMENTS[base]) { + return { box: textBox(latexToUnicode(ctx.wrap(src.slice(start, env.end)))), end: env.end }; + } + let bodyStart = env.bodyStart; + if (base === "alignat" || base === "alignedat" || base === "gatheredat") { + // These carry a required column-count argument `{n}` before the body. + let p = bodyStart; + while (src[p] === " " || src[p] === "\n") p++; + if (src[p] === "{") bodyStart = readBraceGroup(src, p).end; + } + const rows = splitRows(src.slice(bodyStart, env.bodyEnd)) + .map(row => row.trim()) + .filter(row => row !== ""); + if (rows.length === 0) return { box: textBox(""), end: env.end }; + const cellRows = rows.map(splitCells); + let ncols = 0; + for (const row of cellRows) ncols = Math.max(ncols, row.length); + if (ncols <= 1) { + const centered = base === "gather" || base === "gathered" || base === "multline"; + return { + box: vconcat( + rows.map(row => parseExpr(row, ctx)), + centered ? "c" : "l", + ), + end: env.end, + }; + } + // `align`-family semantics: columns alternate right/left in `rl` pairs, a + // thin gap inside each pair and a wide gap between pairs. + const grid = gridBox( + cellRows.map(row => row.map(cell => parseExpr(cell, ctx))), + col => (col % 2 === 0 ? "r" : "l"), + col => (col % 2 === 1 ? 1 : 3), + ); + return { box: grid, end: env.end }; +} + +/** + * Paint every line of `box` through a `latexColorScope` painter so structural + * glyphs (fraction bars, stretched delimiters, matrix brackets) inherit the + * enclosing color scope while nested color runs still restore to it. + */ +function colorizeBox(box: Box, scope: (text: string) => string): Box { + return { lines: box.lines.map(scope), baseline: box.baseline, width: box.width }; +} + +/** + * Parse a math fragment into a layout box. 2-D constructs — fractions, binomials, + * radicals over tall content, `\left…\right` and tall bare parens, environments, + * big-operator limits, block scripts — become stacked boxes; everything between + * them is gathered into inline runs rendered through `latexToUnicode` under the + * active scope wrapper (`ctx`), with `\color` state re-applied per run. + */ +function parseExpr(src: string, ctx: Ctx = ROOT_CTX): Box { const boxes: Box[] = []; let inline = ""; + let color = ""; + let colorScope: ((text: string) => string) | null = null; const flush = (): void => { - if (inline) { - boxes.push(textBox(latexToUnicode(inline))); - inline = ""; - } + if (!inline) return; + boxes.push(textBox(latexToUnicode(ctx.wrap(color + inline)))); + inline = ""; }; + /** Child context carrying the enclosing wrapper plus current color state. */ + const inner = (): Ctx => { + if (!color) return ctx; + const pre = color; + return { wrap: run => ctx.wrap(pre + run) }; + }; + /** Apply the active `\color` scope to a structural box's glyphs. */ + const paint = (box: Box): Box => (colorScope === null ? box : colorizeBox(box, colorScope)); let i = 0; while (i < src.length) { const c = src[i]; @@ -348,19 +866,204 @@ function parseExpr(src: string): Box { flush(); const num = readArg(src, j); const den = readArg(src, num.end); - boxes.push(fracBox(parseExpr(num.text), parseExpr(den.text))); + boxes.push(paint(fracBox(parseExpr(num.text, inner()), parseExpr(den.text, inner())))); i = den.end; continue; } + if (name && BINOM_COMMANDS[name]) { + flush(); + const top = readArg(src, j); + const bottom = readArg(src, top.end); + boxes.push(paint(binomBox(parseExpr(top.text, inner()), parseExpr(bottom.text, inner())))); + i = bottom.end; + continue; + } + if (name === "sqrt") { + let k = j; + while (src[k] === " ") k++; + let degree: string | null = null; + if (src[k] === "[") { + const close = src.indexOf("]", k); + degree = src.slice(k + 1, close === -1 ? src.length : close); + k = close === -1 ? src.length : close + 1; + } + const arg = readArg(src, k); + // Display style always draws the roof (like LaTeX); inline math + // keeps the flat `√(…)` form via latexToUnicode. + flush(); + boxes.push(paint(radicalBox(parseExpr(arg.text, inner()), degree))); + i = arg.end; + continue; + } + if (name === "left") { + const lr = readLeftRight(src, i); + if (lr) { + const segBoxes = lr.segments.map(segment => parseExpr(segment, inner())); + let above = 0; + let below = 0; + for (const b of segBoxes) { + above = Math.max(above, b.baseline); + below = Math.max(below, b.lines.length - 1 - b.baseline); + } + const height = above + below + 1; + if (height === 1) { + // Single-line: keep the whole span inline so converter + // state (fonts, colors, spacing) is preserved. + inline += src.slice(i, lr.end); + i = lr.end; + continue; + } + flush(); + const parts: Box[] = []; + const push = (col: Box | null): void => { + if (col) parts.push(col); + }; + push(delimColumn(delimKey(lr.left), height, above)); + segBoxes.forEach((segment, s) => { + parts.push(segment); + if (s < lr.middles.length) push(delimColumn(delimKey(lr.middles[s]), height, above)); + }); + push(delimColumn(delimKey(lr.right), height, above)); + boxes.push(paint(hconcat(parts))); + i = lr.end; + continue; + } + } + if (name && (LIMIT_OPERATORS[name] || INTEGRAL_OPERATORS[name])) { + let k = j; + while (src[k] === " ") k++; + let stack = LIMIT_OPERATORS[name] === true; + let resume = j; // resume point when the operator stays inline + if (src.startsWith("\\limits", k) && !/[A-Za-z]/.test(src[k + 7] ?? "")) { + stack = true; + resume = k = k + 7; + } else if (src.startsWith("\\nolimits", k) && !/[A-Za-z]/.test(src[k + 9] ?? "")) { + stack = false; + resume = k + 9; + } + if (stack) { + let subText: string | null = null; + let supText: string | null = null; + let m = k; + for (;;) { + // Peek past spaces without consuming them, so a run + // following the operator keeps its leading space. + let n = m; + while (src[n] === " ") n++; + if (src[n] === "_" && subText === null) { + const arg = readArg(src, n + 1); + subText = arg.text; + m = arg.end; + continue; + } + if (src[n] === "^" && supText === null) { + const arg = readArg(src, n + 1); + supText = arg.text; + m = arg.end; + continue; + } + break; + } + if (subText !== null || supText !== null) { + flush(); + const glyph = textBox(latexToUnicode(ctx.wrap(`${color}\\${name}`))); + boxes.push( + paint( + limitsBox( + glyph, + subText === null ? null : parseExpr(subText, inner()), + supText === null ? null : parseExpr(supText, inner()), + ), + ), + ); + i = m; + continue; + } + } + inline += `\\${name}`; + i = resume; + continue; + } + if (name === "color" || name === "normalcolor") { + flush(); // preceding run keeps the previous color + if (name === "normalcolor") { + color = ""; + colorScope = null; + i = j; + continue; + } + let k = j; + while (src[k] === " ") k++; + let opt = ""; + if (src[k] === "[") { + const close = src.indexOf("]", k); + if (close !== -1) { + opt = src.slice(k, close + 1); + k = close + 1; + while (src[k] === " ") k++; + } + } + if (src[k] === "{") { + const spec = readBraceGroup(src, k); + color = `\\color${opt}{${spec.text}}`; + colorScope = latexColorScope(opt ? opt.slice(1, -1).trim() : null, spec.text); + i = spec.end; + } else { + color = ""; + colorScope = null; + i = k; + } + continue; + } if (name === "begin") { - const env = parseEnvironment(src, i); + const env = parseEnvironment(src, i, inner()); if (env) { flush(); - boxes.push(env.box); + boxes.push(paint(env.box)); i = env.end; continue; } } + if (name && (MATH_FONT_COMMANDS.has(name) || name === "textcolor")) { + // Scoped wrapper around 2-D content: recurse with the wrapper + // re-applied to every inline run, so styling crosses boxes. + let k = j; + while (src[k] === " ") k++; + let prefix = `\\${name}`; + let scope: ((text: string) => string) | null = null; + if (name === "textcolor") { + let model: string | null = null; + if (src[k] === "[") { + const close = src.indexOf("]", k); + if (close !== -1) { + model = src.slice(k + 1, close).trim(); + prefix += src.slice(k, close + 1); + k = close + 1; + while (src[k] === " ") k++; + } + } + if (src[k] !== "{") { + inline += `\\${name}`; + i = j; + continue; + } + const spec = readBraceGroup(src, k); + prefix += `{${spec.text}}`; + scope = latexColorScope(model, spec.text); + k = spec.end; + while (src[k] === " ") k++; + } + if (src[k] === "{") { + const content = readBraceGroup(src, k); + flush(); + const pre = color; + let box = parseExpr(content.text, { wrap: run => ctx.wrap(`${pre}${prefix}{${run}}`) }); + if (scope !== null) box = colorizeBox(box, scope); + boxes.push(paint(box)); + i = content.end; + continue; + } + } if (!name) { // Non-letter command (`\\`, `\,`, `\{`, …): keep the 2-char token inline. inline += `\\${src[j] ?? ""}`; @@ -386,18 +1089,72 @@ function parseExpr(src: string): Box { continue; } if (c === "^" || c === "_") { - const script = readScript(src, i); - inline += script.text; - i = script.end; + const first = readScript(src, i); + // Consume an immediately following opposite script (`M_i^j`) so both + // land in one shared column instead of two successive ones. + let second: Span | null = null; + let n = first.end; + while (src[n] === " ") n++; + if (src[n] === (c === "^" ? "_" : "^")) second = readScript(src, n); + const end = second === null ? first.end : second.end; + const supText = c === "^" ? first.text : second?.text; + const subText = c === "_" ? first.text : second?.text; + const supBox = supText === undefined ? null : parseExpr(scriptArgOf(supText), inner()); + const subBox = subText === undefined ? null : parseExpr(scriptArgOf(subText), inner()); + // The converter falls back to `^(…)`/`_(…)` when any character lacks a + // Unicode script form; those scripts get real raised/lowered boxes. + const unconvertible = (raw: string | undefined): boolean => { + if (raw === undefined) return false; + const flat = latexToUnicode(raw); + return flat.startsWith("^") || flat.startsWith("_"); + }; + const tall = (supBox !== null && supBox.lines.length > 1) || (subBox !== null && subBox.lines.length > 1); + if (tall || unconvertible(supText) || unconvertible(subText)) { + // Block script (`x^{\frac{1}{2}}`, `x^q`): raise/lower the boxes + // against the run or box they follow. + flush(); + const base = boxes.pop() ?? textBox(""); + boxes.push(paint(attachScripts(base, subBox, supBox))); + i = end; + continue; + } + const last = boxes[boxes.length - 1]; + if (inline === "" && last !== undefined && last.lines.length > 1) { + // Scripts directly on a tall box (`M^T`, `\right|_{x=a}`): pin + // the Unicode script glyphs (guaranteed convertible here after + // the gate above) to its corners. + const corner = (raw: string | undefined): Box | null => + raw === undefined ? null : textBox(latexToUnicode(ctx.wrap(color + raw))); + boxes[boxes.length - 1] = paint(attachScripts(last, corner(subText), corner(supText))); + i = end; + continue; + } + inline += src.slice(i, end); + i = end; continue; } if (c === "{") { const group = readBraceGroup(src, i); flush(); - boxes.push(parseExpr(group.text)); + boxes.push(paint(parseExpr(group.text, inner()))); i = group.end; continue; } + if (c === "(" || c === "[") { + // Bare delimiters stretch when their content is tall (common in + // model output that omits `\left`/`\right`). + const closeCh = c === "(" ? ")" : "]"; + const close = matchDelim(src, i, c, closeCh); + if (close !== -1) { + const innerBox = parseExpr(src.slice(i + 1, close), inner()); + if (innerBox.lines.length > 1) { + flush(); + boxes.push(paint(delimBox(innerBox, c, closeCh))); + i = close + 1; + continue; + } + } + } inline += c; i++; } @@ -406,7 +1163,7 @@ function parseExpr(src: string): Box { return hconcat(boxes); } -/** Split on top-level `\n` row separators (outside braces and environments). */ +/** Split on top-level `\n` and `\\` row separators (outside braces and environments). */ function splitLines(src: string): string[] { const lines: string[] = []; let braceDepth = 0; @@ -426,7 +1183,18 @@ function splitLines(src: string): string[] { } const c = src[i]; if (c === "\\") { - i += 2; // escaped char / second backslash — never a logical-line break + if (src[i + 1] === "\\" && braceDepth === 0 && envDepth === 0) { + lines.push(src.slice(last, i)); + i += 2; + while (src[i] === " ") i++; + if (src[i] === "[") { + const close = src.indexOf("]", i); + i = close === -1 ? src.length : close + 1; + } + last = i; + continue; + } + i += 2; // escaped char — never a logical-line break continue; } if (c === "{") braceDepth++; @@ -442,10 +1210,11 @@ function splitLines(src: string): string[] { } /** - * Render a display LaTeX math fragment to lines, stacking `\frac` vertically. - * Top-level source newlines become vertical rows (so a `lhs =` line stays above - * its block); each row stacks fractions via `parseExpr`. Inline math should use - * `latexToUnicode` instead — fractions there stay single-line. + * Render a display LaTeX math fragment to lines with full 2-D layout: stacked + * fractions, stretchy delimiters, matrix grids, operator limits, drawn + * radicals. Top-level source newlines and `\\` become vertical rows (so a + * `lhs =` line stays above its block). Inline math should use `latexToUnicode` + * instead — fractions there stay single-line. */ export function latexToBlock(src: string): string[] { if (typeof src !== "string" || src.trim() === "") return []; diff --git a/packages/tui/src/latex-to-unicode.ts b/packages/tui/src/latex-to-unicode.ts index c4c0fcf1d..35a39a93b 100644 --- a/packages/tui/src/latex-to-unicode.ts +++ b/packages/tui/src/latex-to-unicode.ts @@ -317,6 +317,13 @@ const FONTS: Record = { texttt: "mono", textsf: "sans", }; +/** + * Math font command names (`\mathbf`, `\mathbb`, …) whose single brace argument + * restyles glyphs. Exported for the display block engine (`latex-block`), which + * re-wraps inline runs inside these commands when their argument contains 2-D + * layout (fractions, matrices) so styling survives box boundaries. + */ +export const MATH_FONT_COMMANDS: ReadonlySet = new Set(Object.keys(FONTS)); // Text-mode commands whose argument is passed through literally (no math). const TEXT_COMMANDS: Record = { @@ -1136,6 +1143,22 @@ function ansiColor(model: string | null, spec: string): AnsiColor | null { return { foreground, background: foreground.replace("\x1b[38;", "\x1b[48;") }; } +/** + * Painter for a LaTeX color scope (optional model + spec, e.g. `rgb`/`1,0,0` or + * `red`): returns a function that paints already-rendered text with the scope's + * foreground, re-asserting it after embedded foreground resets so nested color + * runs restore to the scope color; null when the color cannot be resolved. Used + * by the display block engine (`latex-block`) to paint structural glyphs + * (fraction bars, stretched delimiters, matrix brackets) inside + * `\color`/`\textcolor` scopes. + */ +export function latexColorScope(model: string | null, spec: string): ((text: string) => string) | null { + const color = ansiColor(model, spec); + if (color === null) return null; + const { foreground } = color; + return text => foreground + text.replaceAll(ANSI_FG_RESET, foreground) + ANSI_FG_RESET; +} + function restoreAnsi( text: string, fromForeground: string | null, diff --git a/packages/tui/test/latex-block.test.ts b/packages/tui/test/latex-block.test.ts index 21e02b03c..2cf3a8168 100644 --- a/packages/tui/test/latex-block.test.ts +++ b/packages/tui/test/latex-block.test.ts @@ -26,8 +26,12 @@ describe("latexToBlock (stacked display fractions)", () => { expect(latexToBlock("\\frac{\\frac{a}{b}}{c}")).toEqual([" a ", " ─── ", " b ", "─────", " c "]); }); - it("keeps a plain expression on a single line", () => { - expect(latexToBlock("e^{i\\pi} + 1 = 0")).toEqual(["e^(iπ) + 1 = 0"]); + it("keeps a fully convertible expression on a single line", () => { + expect(latexToBlock("x^2 + y_1 = 0")).toEqual(["x² + y₁ = 0"]); + }); + + it("raises a non-convertible exponent as a block (Euler's identity)", () => { + expect(latexToBlock("e^{i\\pi} + 1 = 0").map(line => line.trimEnd())).toEqual([" iπ", "e + 1 = 0"]); }); it("stacks fractions inside wrapper environments (equation)", () => { @@ -56,11 +60,19 @@ describe("latexToBlock (stacked display fractions)", () => { expect(stripVTControlCharacters(lines[5])).toContain("4"); }); - it("renders matrices flat (grid environments are not stacked as fractions)", () => { - const lines = latexToBlock("\\begin{bmatrix} a & b \\\\ c & d \\end{bmatrix}"); - expect(lines.length).toBe(2); - expect(lines[0].startsWith("[")).toBe(true); - expect(lines[lines.length - 1].endsWith("]")).toBe(true); + it("renders matrix environments as center-baselined grids in stretched brackets", () => { + expect(latexToBlock("\\begin{bmatrix} a & b \\\\ c & d \\end{bmatrix}")).toEqual([ + "⎡ a b ⎤", + "⎢ ⎥", + "⎣ c d ⎦", + ]); + expect(latexToBlock("\\begin{pmatrix} a & b \\\\ c & d \\end{pmatrix}")).toEqual([ + "⎛ a b ⎞", + "⎜ ⎟", + "⎝ c d ⎠", + ]); + // Single-row matrices stay flat. + expect(latexToBlock("\\begin{pmatrix} a & b & c \\end{pmatrix}")).toEqual(["(a b c)"]); }); it("centers using visible width, ignoring ANSI color codes in a numerator", () => { @@ -77,3 +89,161 @@ describe("latexToBlock (stacked display fractions)", () => { expect(latexToBlock(" ")).toEqual([]); }); }); +describe("latexToBlock (2-D layout)", () => { + it("baseline-aligns matrix cells containing fractions", () => { + expect(latexToBlock("\\begin{bmatrix} \\frac{1}{2} & x \\\\ y & z \\end{bmatrix}")).toEqual([ + "⎡ 1 ⎤", + "⎢ ─── x ⎥", + "⎢ 2 ⎥", + "⎢ ⎥", + "⎣ y z ⎦", + ]); + }); + + it("centers surrounding text on the matrix middle", () => { + expect(latexToBlock("A = \\begin{bmatrix} a \\\\ b \\end{bmatrix}")).toEqual([ + " ⎡ a ⎤", + "A = ⎢ ⎥", + " ⎣ b ⎦", + ]); + }); + + it("renders vmatrix with full-height bars", () => { + expect(latexToBlock("\\begin{vmatrix} a & b \\\\ c & d \\end{vmatrix}")).toEqual([ + "│ a b │", + "│ │", + "│ c d │", + ]); + }); + + it("honors the array column specification", () => { + expect( + latexToBlock("\\begin{array}{lcr} 1 & 22 & 333 \\\\ aaa & b & c \\end{array}").map(line => line.trimEnd()), + ).toEqual(["1 22 333", "", "aaa b c"]); + }); + + it("renders cases with a stretched left brace and left-aligned columns", () => { + const lines = latexToBlock("f(x) = \\begin{cases} x & x > 0 \\\\ 0 & \\text{otherwise} \\end{cases}"); + expect(lines.map(line => line.trimEnd())).toEqual([" ⎧ x x > 0", "f(x) = ⎨", " ⎩ 0 otherwise"]); + }); + + it("stacks big-operator limits above and below the symbol", () => { + expect(latexToBlock("\\sum_{i=0}^{n} i^2")).toEqual([" n ", " ∑ i²", "i=0 "]); + }); + + it("places \\lim scripts underneath", () => { + const lines = latexToBlock("\\lim_{x \\to 0} \\frac{\\sin x}{x}"); + expect(lines[1]).toContain("lim"); + expect(lines[2]).toContain("x → 0"); + expect(lines[1]).toContain("───"); // fraction bar on the lim baseline row + }); + + it("keeps integral bounds beside the symbol unless \\limits is given", () => { + expect(latexToBlock("\\int_a^b f(x) dx")).toEqual(["∫ₐᵇ f(x) dx"]); + expect(latexToBlock("\\int\\limits_a^b f(x) dx").map(line => line.trimEnd())).toEqual(["b", "∫ f(x) dx", "a"]); + }); + + it("stretches \\left…\\right delimiters around tall content and pins corner scripts", () => { + expect(latexToBlock("\\left( \\frac{a+b}{c} \\right)^2").map(line => line.trimEnd())).toEqual([ + "⎛ a+b ⎞²", + "⎜ ───── ⎟", + "⎝ c ⎠", + ]); + }); + + it("stretches bare parentheses around a fraction", () => { + expect(latexToBlock("( \\frac{a}{b} )")).toEqual(["⎛ a ⎞", "⎜ ─── ⎟", "⎝ b ⎠"]); + }); + + it("leaves unbalanced interval brackets on the baseline", () => { + expect(latexToBlock("[0, 1)")).toEqual(["[0, 1)"]); + }); + + it("renders \\middle delimiters at full height inside \\left…\\right", () => { + const lines = latexToBlock("\\left\\{ x \\middle| \\frac{x}{2} \\in \\mathbb{Z} \\right\\}"); + expect(lines.length).toBe(3); + expect(lines[1].startsWith("⎨")).toBe(true); + expect(lines[1]).toContain("│"); + expect(lines[1].endsWith("⎬")).toBe(true); + }); + + it("always draws the radical roof in display math", () => { + expect(latexToBlock("\\sqrt{\\frac{a+1}{b}}").map(line => line.trimEnd())).toEqual([ + " ┌──────", + " │ a+1", + " │ ─────", + "╲│ b", + ]); + expect(latexToBlock("\\sqrt{x}").map(line => line.trimEnd())).toEqual([" ┌──", "╲│ x"]); + }); + + it("stacks \\binom inside stretched parentheses", () => { + expect(latexToBlock("\\binom{n}{k}")).toEqual(["⎛ n ⎞", "⎜ ⎟", "⎝ k ⎠"]); + }); + + it("raises a block superscript containing a fraction", () => { + expect(latexToBlock("e^{\\frac{x}{2}}").map(line => line.trimEnd())).toEqual([" x", " ───", " 2", "e"]); + }); + it("raises/lowers one-line scripts that have no Unicode form", () => { + // `q` has no superscript/subscript code point; a real box replaces `^(q)`. + expect(latexToBlock("x^q").map(line => line.trimEnd())).toEqual([" q", "x"]); + expect(latexToBlock("x_q").map(line => line.trimEnd())).toEqual(["x", " q"]); + expect(latexToBlock("x_q^q").map(line => line.trimEnd())).toEqual([" q", "x", " q"]); + }); + + it("pins both scripts of a tall base in one shared column", () => { + expect(latexToBlock("\\begin{bmatrix} a & b \\\\ c & d \\end{bmatrix}_0^T").map(line => line.trimEnd())).toEqual([ + "⎡ a b ⎤ᵀ", + "⎢ ⎥", + "⎣ c d ⎦₀", + ]); + }); + + it("aligns align-environment rows on the & column", () => { + expect( + latexToBlock("\\begin{align} f(x) &= x^2 + 1 \\\\ g(x) &= \\frac{x}{2} \\end{align}").map(line => + line.trimEnd(), + ), + ).toEqual(["f(x) = x² + 1", " x", "g(x) = ───", " 2"]); + }); + + it("centers gather-environment rows", () => { + expect(latexToBlock("\\begin{gather} a = b \\\\ longer = expression \\end{gather}")).toEqual([ + " a = b ", + "longer = expression", + ]); + }); + + it("splits top-level \\\\ into vertical rows", () => { + expect(latexToBlock("a \\\\ b")).toEqual(["a", "b"]); + }); + + it("keeps \\color scope across a stacked fraction, painting the bar", () => { + Object.assign(TERMINAL, { trueColor: true }); + const lines = latexToBlock("\\color{red} x + \\frac{a}{b}"); + expect(lines.map(stripVTControlCharacters).map(line => line.trimEnd())).toEqual([ + " a", + " x + ───", + " b", + ]); + expect(lines[0]).toContain("\x1b[38;"); // numerator run is colored + expect(lines[1]).toContain("\x1b[38;"); // "x + " run and the bar are colored + }); + + it("paints textcolor-scoped structural glyphs (fraction bar)", () => { + Object.assign(TERMINAL, { trueColor: true }); + const lines = latexToBlock("\\textcolor{red}{\\frac{a}{b}}"); + expect(lines.map(stripVTControlCharacters)).toEqual([" a ", "───", " b "]); + expect(lines[1]).toContain("\x1b[38;"); // the synthesized bar inherits the scope color + }); + + it("styles fonts across a stacked fraction (\\mathbf)", () => { + expect(latexToBlock("\\mathbf{\\frac{a}{b}}")).toEqual([" 𝐚 ", "───", " 𝐛 "]); + }); + + it("stacks limit operators inside styling wrappers", () => { + Object.assign(TERMINAL, { trueColor: true }); + const lines = latexToBlock("\\textcolor{red}{\\sum_{i=1}^n}"); + expect(lines.map(stripVTControlCharacters).map(line => line.trimEnd())).toEqual([" n", " ∑", "i=1"]); + }); +}); diff --git a/packages/tui/test/markdown-math.test.ts b/packages/tui/test/markdown-math.test.ts index b80dce53d..e302feadd 100644 --- a/packages/tui/test/markdown-math.test.ts +++ b/packages/tui/test/markdown-math.test.ts @@ -23,21 +23,22 @@ describe("Markdown math rendering", () => { expect(line).toBe("energy xᵢ² + yⱼ² done"); }); - it("renders an own-line $$…$$ matrix block across multiple lines", () => { + it("renders an own-line $$…$$ matrix block as a bracketed grid", () => { const lines = renderLines("$$\n\\begin{bmatrix} a & b \\\\ c & d \\end{bmatrix}\n$$"); - // Two rows, not collapsed onto one line. - expect(lines.length).toBe(2); - expect(lines[0].startsWith("[")).toBe(true); - expect(lines[lines.length - 1].endsWith("]")).toBe(true); - expect(lines.join("").replace(/[\s[\]]/g, "")).toBe("abcd"); + // Two content rows around a centering gap row, in stretched brackets. + expect(lines.length).toBe(3); + expect(lines[0].startsWith("⎡")).toBe(true); + expect(lines[lines.length - 1].endsWith("⎦")).toBe(true); + expect(lines.join("").replace(/[\s⎡⎤⎢⎥⎣⎦]/g, "")).toBe("abcd"); }); - it("stacks a \\[…\\] display fraction (quadratic formula)", () => { + it("stacks a \\[…\\] display quadratic formula with a drawn radical", () => { const lines = renderLines("\\[\nx = \\frac{-b \\pm \\sqrt{b^2 - 4ac}}{2a}\n\\]"); - const barRow = lines.findIndex(line => line.includes("─")); - expect(barRow).toBeGreaterThan(0); - expect(lines[barRow]).toContain("x ="); - expect(lines[barRow - 1]).toContain("-b ± √(b² - 4ac)"); + const barRow = lines.findIndex(line => line.includes("x =")); + expect(barRow).toBeGreaterThan(1); + expect(lines[barRow]).toContain("───"); + expect(lines[barRow - 1]).toContain("-b ± ╲│ b² - 4ac"); + expect(lines[barRow - 2]).toContain("┌"); expect(lines[barRow + 1]).toContain("2a"); }); @@ -53,8 +54,8 @@ describe("Markdown math rendering", () => { it("keeps display math inside a list item multi-line", () => { const lines = renderLines("- result:\n\n $$\n \\begin{bmatrix} a \\\\ b \\end{bmatrix}\n $$"); // The matrix rows must land on distinct lines (not flattened to "[a b]"). - const openRow = lines.findIndex(line => line.includes("[a")); - const closeRow = lines.findIndex(line => line.includes("b]")); + const openRow = lines.findIndex(line => line.includes("⎡ a")); + const closeRow = lines.findIndex(line => line.includes("b ⎦")); expect(openRow).toBeGreaterThanOrEqual(0); expect(closeRow).toBeGreaterThan(openRow); }); diff --git a/packages/tui/test/markdown.test.ts b/packages/tui/test/markdown.test.ts index dc33d4420..233c15b6e 100644 --- a/packages/tui/test/markdown.test.ts +++ b/packages/tui/test/markdown.test.ts @@ -1830,11 +1830,11 @@ describe("Math rendering", () => { expect(out).not.toContain("begin{cases}"); }); - it("converts a $$-delimited matrix block to multi-line Unicode", () => { + it("converts a $$-delimited matrix block to a parenthesized grid", () => { const md = new Markdown("$$\n\\begin{pmatrix} a & b \\\\ c & d \\end{pmatrix}\n$$", 0, 0, defaultMarkdownTheme); const out = plain(md); - expect(out).toContain("(a"); - expect(out).toContain("d)"); + expect(out).toContain("⎛ a"); + expect(out).toContain("d ⎠"); expect(out).not.toContain("pmatrix"); }); From 395e4a5fcfceb0802e130370e358573636c2aad8 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 10 Jul 2026 20:50:59 +0200 Subject: [PATCH 088/205] chore: bump version to 16.4.1 --- Cargo.lock | 10 +-- Cargo.toml | 2 +- bun.lock | 92 +++++++++++++++------------ crates/pi-natives/src/lib.rs | 2 +- package.json | 24 +++---- packages/agent/CHANGELOG.md | 2 + packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 + packages/ai/package.json | 2 +- packages/catalog/CHANGELOG.md | 2 + packages/catalog/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 + packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 25 files changed, 96 insertions(+), 74 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 2fdf19aee..1bff4d104 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2881,7 +2881,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.4.0" +version = "16.4.1" dependencies = [ "anyhow", "ast-grep-core", @@ -2950,7 +2950,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.4.0" +version = "16.4.1" dependencies = [ "async-trait", "libc", @@ -2962,7 +2962,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.4.0" +version = "16.4.1" dependencies = [ "anyhow", "arboard", @@ -3015,7 +3015,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.4.0" +version = "16.4.1" dependencies = [ "anyhow", "brush-builtins", @@ -3064,7 +3064,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.4.0" +version = "16.4.1" dependencies = [ "dashmap", "globset", diff --git a/Cargo.toml b/Cargo.toml index 0e637b62a..daa3c6881 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.4.0" +version = "16.4.1" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index 0cae7fb8e..0395e5c66 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.4.0", + "version": "16.4.1", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.4.0", + "version": "16.4.1", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.4.0", + "version": "16.4.1", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.4.0", + "version": "16.4.1", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.4.0", + "version": "16.4.1", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.4.0", + "version": "16.4.1", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.4.0", + "version": "16.4.1", "devDependencies": { "@types/bun": "catalog:", }, @@ -338,18 +338,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.4.0", - "@oh-my-pi/omp-stats": "16.4.0", - "@oh-my-pi/pi-agent-core": "16.4.0", - "@oh-my-pi/pi-ai": "16.4.0", - "@oh-my-pi/pi-catalog": "16.4.0", - "@oh-my-pi/pi-coding-agent": "16.4.0", - "@oh-my-pi/pi-mnemopi": "16.4.0", - "@oh-my-pi/pi-natives": "16.4.0", - "@oh-my-pi/pi-tui": "16.4.0", - "@oh-my-pi/pi-utils": "16.4.0", - "@oh-my-pi/pi-wire": "16.4.0", - "@oh-my-pi/snapcompact": "16.4.0", + "@oh-my-pi/hashline": "16.4.1", + "@oh-my-pi/omp-stats": "16.4.1", + "@oh-my-pi/pi-agent-core": "16.4.1", + "@oh-my-pi/pi-ai": "16.4.1", + "@oh-my-pi/pi-catalog": "16.4.1", + "@oh-my-pi/pi-coding-agent": "16.4.1", + "@oh-my-pi/pi-mnemopi": "16.4.1", + "@oh-my-pi/pi-natives": "16.4.1", + "@oh-my-pi/pi-tui": "16.4.1", + "@oh-my-pi/pi-utils": "16.4.1", + "@oh-my-pi/pi-wire": "16.4.1", + "@oh-my-pi/snapcompact": "16.4.1", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -485,7 +485,7 @@ "@emnapi/core": ["@emnapi/core@1.11.1", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" } }, "sha512-RSvbQmHzdKzNsLYa/wHrbc3KN4sYLKAdPZxqiM2HATqv/SBk2/ENSHpvXGaLOMcsAyz0poEGqkmmKYG3OWiJEQ=="], - "@emnapi/runtime": ["@emnapi/runtime@1.11.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-kyOl3X0DuTiT1h2ft8r2fYO8JYtU9a9Xis/zBSiGArNaagCOWx90N1k2wxp18czFDH+OgcWGb5ZP/XMt3dcyPA=="], + "@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], "@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="], @@ -957,7 +957,7 @@ "bluebird": ["bluebird@3.4.7", "", {}, "sha512-iD3898SR7sWVRHbiQv+sHUtHnMvC1o3nW5rAcqnq3uOn07DSAppZYUkIGslDz6gXC7HfunPe7YVBgoEJASPcHA=="], - "boolbase": ["boolbase@1.0.0", "", {}, "sha512-JZOSA7Mo9sNGB8+UjSgzdLtokWAky1zbztM3WRLCbZ70/3cTANmQmOdR7y2g+J0e2WXywy1yS468tY+IruqEww=="], + "boolbase": ["boolbase@2.0.0", "", {}, "sha512-DkVaaQHymRhpYEYo9x1oo7Q7B0Y6KJUsjm3c9eTyFDby4MHLBTwZ6ZDWBel5zrYxj1WsZgC5oLpiz+93MluXeA=="], "boolean": ["boolean@3.2.0", "", {}, "sha512-d0II/GO9uf9lfUHH2BQsjxzRJZBdsjgsBiW4BvhWk/3qoKwQFjIDVN19PfX8F2D/r9PCMTtLWjYVCFrpeYUzsw=="], @@ -1009,9 +1009,9 @@ "core-util-is": ["core-util-is@1.0.3", "", {}, "sha512-ZQBvi1DcpJ4GDqanjucZ2Hj3wEO5pZDS89BWbkcrvdxksJorwUDDZamX9ldFkp9aw2lmBDLgkObEA4DWNJ9FYQ=="], - "css-select": ["css-select@5.2.2", "", { "dependencies": { "boolbase": "^1.0.0", "css-what": "^6.1.0", "domhandler": "^5.0.2", "domutils": "^3.0.1", "nth-check": "^2.0.1" } }, "sha512-TizTzUddG/xYLA3NXodFM0fSbNizXjOKhqiQQwvhlspadZokn1KDy0NZFS0wuEubIYAV5/c1/lAr0TaaFXEXzw=="], + "css-select": ["css-select@7.0.0", "", { "dependencies": { "boolbase": "^2.0.0", "css-what": "^8.0.0", "domhandler": "^6.0.1", "domutils": "^4.0.2", "nth-check": "^3.0.1" } }, "sha512-snmjEVXy+1LnwXdxhYvTMj1d9tOh4HxkA1YmoayVBeeyR2C14Pum7fcxJIm4SswYspVy866eYNwlH6xC3/VH5g=="], - "css-what": ["css-what@6.2.2", "", {}, "sha512-u/O3vwbptzhMs3L1fQE82ZSLHQQfto5gyZzwteVIEyeaY5Fc7R4dapF/BvRoSYFeqfBk4m0V1Vafq5Pjv25wvA=="], + "css-what": ["css-what@8.0.0", "", {}, "sha512-DH0Bqq3DNp5tdOReuNyAA+Ev4Y2GS5FMbZpeTLP6C4CDi0h5nL0BmUPChXw3o/qbHLDWHl49sbNqQVY7bMSDdw=="], "cssom": ["cssom@0.5.0", "", {}, "sha512-iKuQcq+NdHqlAcwUY0o/HL69XQrUaQdMjmStJ8JFmUaiiQErlhrmuigkg/CU4E2J0IyUKUrMAgl36TvN67MqTw=="], @@ -1035,13 +1035,13 @@ "dingbat-to-unicode": ["dingbat-to-unicode@1.0.1", "", {}, "sha512-98l0sW87ZT58pU4i61wa2OHwxbiYSbuxsCBozaVnYX2iCnr3bLM3fIes1/ej7h1YdOKuKt/MLs706TVnALA65w=="], - "dom-serializer": ["dom-serializer@2.0.0", "", { "dependencies": { "domelementtype": "^2.3.0", "domhandler": "^5.0.2", "entities": "^4.2.0" } }, "sha512-wIkAryiqt/nV5EQKqQpo3SToSOV9J0DnbJqwK7Wv/Trc92zIAYZ4FlMu+JPFW1DfGFt81ZTCGgDEabffXeLyJg=="], + "dom-serializer": ["dom-serializer@3.1.1", "", { "dependencies": { "domelementtype": "^3.0.0", "domhandler": "^6.0.0", "entities": "^8.0.0" } }, "sha512-4MEa38/QexBob6gFNwu+EGdWvhJ1OKuNwdYY3Y3NyeWDQfnGeDYQUDfIRzWu5B5gsv03so2Uxd28YC6zrsx3Lw=="], "domelementtype": ["domelementtype@2.3.0", "", {}, "sha512-OLETBj6w0OsagBwdXnPdN0cnMfF9opN69co+7ZrbfPGrdpPVNBUj02spi6B1N7wChLQiPn4CSH/zJvXw56gmHw=="], - "domhandler": ["domhandler@5.0.3", "", { "dependencies": { "domelementtype": "^2.3.0" } }, "sha512-cgwlv/1iFQiFnU96XXgROh8xTeetsnJiDsTc7TYCLFd9+/WNkIqPTxiM/8pSd8VIrhXGTf1Ny1q1hquVqDJB5w=="], + "domhandler": ["domhandler@6.0.1", "", { "dependencies": { "domelementtype": "^3.0.0" } }, "sha512-gYzvtM72ZtxQO0T048kd6HWSbbGCNOUwcnfQ01cqIJ4X2IYKFFHZ5mKvrQETcFXxsRObZulDaKmy//R7TPtsBg=="], - "domutils": ["domutils@3.2.2", "", { "dependencies": { "dom-serializer": "^2.0.0", "domelementtype": "^2.3.0", "domhandler": "^5.0.3" } }, "sha512-6kZKyUajlDuqlHKVX1w7gyslj9MPIXzIFiz/rGu35uC1wMi+kMhQwGhl4lt9unC9Vb9INnY9Z3/ZA3+FhASLaw=="], + "domutils": ["domutils@4.0.2", "", { "dependencies": { "dom-serializer": "^3.0.0", "domelementtype": "^3.0.0", "domhandler": "^6.0.0" } }, "sha512-qI4JLRKnSzqFqr7hAlS5xQDusBCjKSEG4t4+7aNrIQMHBcsC2TGEhuyABJdYkgSewL57PNLYEiibY2iPKhKpaA=="], "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], @@ -1189,7 +1189,7 @@ "lightningcss-win32-x64-msvc": ["lightningcss-win32-x64-msvc@1.32.0", "", { "os": "win32", "cpu": "x64" }, "sha512-Amq9B/SoZYdDi1kFrojnoqPLxYhQ4Wo5XiL8EVJrVsB8ARoC1PWW6VGtT0WKCemjy8aC+louJnjS7U18x3b06Q=="], - "linkedom": ["linkedom@0.18.12", "", { "dependencies": { "css-select": "^5.1.0", "cssom": "^0.5.0", "html-escaper": "^3.0.3", "htmlparser2": "^10.0.0", "uhyphen": "^0.2.0" }, "peerDependencies": { "canvas": ">= 2" }, "optionalPeers": ["canvas"] }, "sha512-jalJsOwIKuQJSeTvsgzPe9iJzyfVaEJiEXl+25EkKevsULHvMJzpNqwvj1jOESWdmgKDiXObyjOYwlUqG7wo1Q=="], + "linkedom": ["linkedom@0.18.13", "", { "dependencies": { "css-select": "^7.0.0", "cssom": "^0.5.0", "html-escaper": "^3.0.3", "htmlparser2": "^10.1.0", "uhyphen": "^0.2.0" }, "peerDependencies": { "canvas": ">= 2" }, "optionalPeers": ["canvas"] }, "sha512-ES/o9qotMpzpN2MHs+Iq/JcVoOj8Fa5wiQYrTdFpvAnwXL0g66XHHUc9WUMk6nAlBtGsFQ24ne+SYnvnaQ2FSw=="], "lint-staged": ["lint-staged@17.0.8", "", { "dependencies": { "listr2": "^10.2.1", "picomatch": "^4.0.4", "string-argv": "^0.3.2", "tinyexec": "^1.2.4" }, "optionalDependencies": { "yaml": "^2.9.0" }, "bin": { "lint-staged": "bin/lint-staged.js" } }, "sha512-B2P/d+jVW0UXOQ0MVMLrB/9ydA1P+zz6jYfdrbbEd9ur3S2rcbduFWKiUCC02Sm5hbC8nrm7y24WuYMG54HfxA=="], @@ -1247,7 +1247,7 @@ "node-releases": ["node-releases@2.0.50", "", {}, "sha512-J6l92tKHX6w8Jy5nO1Vuc01NoIiRGi/d6qBKVxh+IQ8Cr3b6HbVNfKiF8ZpFKufTwpwxMmce2W3iQZ861ZRyTg=="], - "nth-check": ["nth-check@2.1.1", "", { "dependencies": { "boolbase": "^1.0.0" } }, "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w=="], + "nth-check": ["nth-check@3.0.1", "", { "dependencies": { "boolbase": "^2.0.0" } }, "sha512-GX0gsdbGVCgnRgbeGaubfjpBXyYRWOOCVeYh08bSQvDZqxz5ndXs1OTfAt/h36G1xvI94YIspsI0sVFqAV9+RQ=="], "object-hash": ["object-hash@3.0.0", "", {}, "sha512-RSn9F68PjH9HqtltsSnqYC1XXoWe9Bju5+213R98cNGttag9q9yAOTzdbsqvIa7aNm5WffBZFpWYr2aWrklWAw=="], @@ -1343,7 +1343,7 @@ "sherpa-onnx-darwin-x64": ["sherpa-onnx-darwin-x64@1.13.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-6RGeis9K9gV/UQWOgd6Rf3iqXr2/YsBQswxHaCR4hrYkHfEIpHMfFmRWLt6nJJCOWgYW2xFxEd9yzjrafAV/Pw=="], - "sherpa-onnx-linux-arm64": ["sherpa-onnx-linux-arm64@1.13.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-uDtZkkoP6QQ/3DHOscCpEZ2WpaiHUQsDpbyYaHURrJ7DbsjqGnS6G8l+R589Ro5Bf282QElzBy3okwxXbt3Kxw=="], + "sherpa-onnx-linux-arm64": ["sherpa-onnx-linux-arm64@1.13.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-RMjMRqT82BgTXypNNGmLe6ZFYhc3WEvnAGl3DdkK7qB/kuXwkL3iHhV31wAecbnWPsnEpUoD+8cFovWSBzsCuw=="], "sherpa-onnx-linux-x64": ["sherpa-onnx-linux-x64@1.13.4", "", { "os": "linux", "cpu": "x64" }, "sha512-WZh5NCkGPFHHpYSd78iN4OnmxQeSTGyt9uZskH+im/NFHQ7elQ7B0sLzCMeRpvJxiIKvd9C6WxIJ4hYaxClfsQ=="], @@ -1493,11 +1493,9 @@ "@opentelemetry/sdk-metrics/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - "@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], + "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.1", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-RSvbQmHzdKzNsLYa/wHrbc3KN4sYLKAdPZxqiM2HATqv/SBk2/ENSHpvXGaLOMcsAyz0poEGqkmmKYG3OWiJEQ=="], - "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], - - "@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.2", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-kyOl3X0DuTiT1h2ft8r2fYO8JYtU9a9Xis/zBSiGArNaagCOWx90N1k2wxp18czFDH+OgcWGb5ZP/XMt3dcyPA=="], + "@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], "@tailwindcss/oxide-wasm32-wasi/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="], @@ -1517,12 +1515,22 @@ "cliui/wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="], - "dom-serializer/entities": ["entities@4.5.0", "", {}, "sha512-V0hjH4dGPh9Ao5p0MoRY6BVqtwCjhz6vI5LT8AJ55H+4g9/4vbHx1I54fS0XuclLhDHArPQCiMjDxjaL8fPxhw=="], + "dom-serializer/domelementtype": ["domelementtype@3.0.0", "", {}, "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg=="], + + "dom-serializer/entities": ["entities@8.0.0", "", {}, "sha512-zwfzJecQ/Uej6tusMqwAqU/6KL2XaB2VZ2Jg54Je6ahNBGNH6Ek6g3jjNCF0fG9EWQKGZNddNjU5F1ZQn/sBnA=="], + + "domhandler/domelementtype": ["domelementtype@3.0.0", "", {}, "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg=="], + + "domutils/domelementtype": ["domelementtype@3.0.0", "", {}, "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg=="], "fastembed/onnxruntime-node": ["onnxruntime-node@1.21.0", "", { "dependencies": { "global-agent": "^3.0.0", "onnxruntime-common": "1.21.0", "tar": "^7.0.1" }, "os": [ "linux", "win32", "darwin", ] }, "sha512-NeaCX6WW2L8cRCSqy3bInlo5ojjQqu2fD3D+9W5qb5irwxhEyWKXeH2vZ8W9r6VxaMPUan+4/7NDwZMtouZxEw=="], "fs-minipass/minipass": ["minipass@3.3.6", "", { "dependencies": { "yallist": "^4.0.0" } }, "sha512-DxiNidxSEK+tHG6zOIklvNOwm3hvCrbUrdtzY74U6HKTJxvIDfOUL5W5P2Ghd3DTkhhKPYGqeNUIh5qcM4YBfw=="], + "htmlparser2/domhandler": ["domhandler@5.0.3", "", { "dependencies": { "domelementtype": "^2.3.0" } }, "sha512-cgwlv/1iFQiFnU96XXgROh8xTeetsnJiDsTc7TYCLFd9+/WNkIqPTxiM/8pSd8VIrhXGTf1Ny1q1hquVqDJB5w=="], + + "htmlparser2/domutils": ["domutils@3.2.2", "", { "dependencies": { "dom-serializer": "^2.0.0", "domelementtype": "^2.3.0", "domhandler": "^5.0.3" } }, "sha512-6kZKyUajlDuqlHKVX1w7gyslj9MPIXzIFiz/rGu35uC1wMi+kMhQwGhl4lt9unC9Vb9INnY9Z3/ZA3+FhASLaw=="], + "js-yaml/argparse": ["argparse@2.0.1", "", {}, "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q=="], "jszip/readable-stream": ["readable-stream@2.3.8", "", { "dependencies": { "core-util-is": "~1.0.0", "inherits": "~2.0.3", "isarray": "~1.0.0", "process-nextick-args": "~2.0.0", "safe-buffer": "~5.1.1", "string_decoder": "~1.1.1", "util-deprecate": "~1.0.1" } }, "sha512-8p0AUk4XODgIewSi0l8Epjs+EVnWiK7NoDIEGU0HhE7+ZyY8D1IMY7odu5lRrFXGg71L15KG8QrPmum45RTtdA=="], @@ -1561,6 +1569,8 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.19", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-4LeEWl96twnS2Q7Bz4MGqgazLqO+hJN63GZxXoIqh1T3VweYD997gbU1ItNsQafqqXTXd5WFyFdReLtwvRBNiw=="], + "htmlparser2/domutils/dom-serializer": ["dom-serializer@2.0.0", "", { "dependencies": { "domelementtype": "^2.3.0", "domhandler": "^5.0.2", "entities": "^4.2.0" } }, "sha512-wIkAryiqt/nV5EQKqQpo3SToSOV9J0DnbJqwK7Wv/Trc92zIAYZ4FlMu+JPFW1DfGFt81ZTCGgDEabffXeLyJg=="], + "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], "@huggingface/transformers/onnxruntime-node/global-agent/matcher": ["matcher@3.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng=="], @@ -1581,6 +1591,8 @@ "fastembed/onnxruntime-node/tar/yallist": ["yallist@5.0.0", "", {}, "sha512-YgvUTfwqyc7UXVMrB+SImsVYSmTS8X/tSrtdNZMImM+n7+QTriRXyXim0mBrTXNeqzVF0KWGgHPeiyViFFrNDw=="], + "htmlparser2/domutils/dom-serializer/entities": ["entities@4.5.0", "", {}, "sha512-V0hjH4dGPh9Ao5p0MoRY6BVqtwCjhz6vI5LT8AJ55H+4g9/4vbHx1I54fS0XuclLhDHArPQCiMjDxjaL8fPxhw=="], + "@huggingface/transformers/onnxruntime-node/global-agent/serialize-error/type-fest": ["type-fest@0.13.1", "", {}, "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg=="], "fastembed/onnxruntime-node/global-agent/serialize-error/type-fest": ["type-fest@0.13.1", "", {}, "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 0ffc82553..2709c532d 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_4_0")] +#[napi(js_name = "__piNativesV16_4_1")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index 0d448221b..bfaee536f 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.4.0", - "@oh-my-pi/omp-stats": "16.4.0", - "@oh-my-pi/pi-agent-core": "16.4.0", - "@oh-my-pi/pi-ai": "16.4.0", - "@oh-my-pi/pi-catalog": "16.4.0", - "@oh-my-pi/pi-coding-agent": "16.4.0", - "@oh-my-pi/pi-mnemopi": "16.4.0", - "@oh-my-pi/pi-natives": "16.4.0", - "@oh-my-pi/pi-tui": "16.4.0", - "@oh-my-pi/pi-utils": "16.4.0", - "@oh-my-pi/pi-wire": "16.4.0", - "@oh-my-pi/snapcompact": "16.4.0", + "@oh-my-pi/hashline": "16.4.1", + "@oh-my-pi/omp-stats": "16.4.1", + "@oh-my-pi/pi-agent-core": "16.4.1", + "@oh-my-pi/pi-ai": "16.4.1", + "@oh-my-pi/pi-catalog": "16.4.1", + "@oh-my-pi/pi-coding-agent": "16.4.1", + "@oh-my-pi/pi-mnemopi": "16.4.1", + "@oh-my-pi/pi-natives": "16.4.1", + "@oh-my-pi/pi-tui": "16.4.1", + "@oh-my-pi/pi-utils": "16.4.1", + "@oh-my-pi/pi-wire": "16.4.1", + "@oh-my-pi/snapcompact": "16.4.1", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index e0989db65..568e7f2b0 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.1] - 2026-07-10 + ### Fixed - Enabled reasoning encryption content for all Responses Lite compaction requests diff --git a/packages/agent/package.json b/packages/agent/package.json index d87c298a5..9e561e191 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.4.0", + "version": "16.4.1", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 8b65af82f..4dbaf5b00 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.1] - 2026-07-10 + ### Changed - Enforced `all_turns` reasoning context for all Responses Lite requests diff --git a/packages/ai/package.json b/packages/ai/package.json index 634515350..54033346b 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.4.0", + "version": "16.4.1", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index d7efd673c..4d9cfbdd6 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.1] - 2026-07-10 + ### Added - Added GPT-5.6 Luna, Sol, and Terra models diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 1ad9e3356..0ee74b89d 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.4.0", + "version": "16.4.1", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6e6c767c2..9ae1ccbd7 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.1] - 2026-07-10 + ### Changed - Reduced agent bias against large diffs and refactors in advisor prompts diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index d64ab8d54..fcf13ea2d 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.4.0", + "version": "16.4.1", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 74679635f..2218fcc40 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.4.0", + "version": "16.4.1", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index a6a86829d..8ca2d36d3 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.4.0", + "version": "16.4.1", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 4da2cdc32..2f377dd9c 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -170,7 +170,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_4_0(): void +export declare function __piNativesV16_4_1(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 3b6b45f4b..02486bec8 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_4_0 = nativeBindings.__piNativesV16_4_0; +export const __piNativesV16_4_1 = nativeBindings.__piNativesV16_4_1; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 657ac54bd..a4babd46f 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.4.0", + "version": "16.4.1", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index a2df06664..2b5292b64 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.4.0", + "version": "16.4.1", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index faa1c8032..0d8293ab6 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.4.0", + "version": "16.4.1", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 693be1114..49d8bc48a 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.4.0", + "version": "16.4.1", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 7e6b2eb9e..5d1a9980b 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.1] - 2026-07-10 + ### Added - Added full 2-D layout support for display LaTeX math (fractions, matrices, radicals, limits), modeled on the layout approach of [txm](https://github.com/thatmagicalcat/txm) (Terminal TeX Math) by [@thatmagicalcat](https://github.com/thatmagicalcat) diff --git a/packages/tui/package.json b/packages/tui/package.json index 02a414325..cd2c1025d 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.4.0", + "version": "16.4.1", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 1bbeb5e2f..c4be6563f 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.4.0", + "version": "16.4.1", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index e57be9695..1191a7bd2 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.4.0", + "version": "16.4.1", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From 9b4dcaa1140bfc9eea44cf4e1057fed3257b2b0f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 15:24:27 -0300 Subject: [PATCH 089/205] fix(ai): adapted xai responses replay shapes Convert freeform custom_tool_call history to function_call pairs and clamp input_image.detail original to auto when replaying into xAI OAuth Responses, so session continuations stop 422ing. Fixes #5002 --- packages/ai/CHANGELOG.md | 3 + packages/ai/src/providers/openai-shared.ts | 84 +++++- packages/ai/src/utils.ts | 51 +++- .../openai-responses-history-payload.test.ts | 257 +++++++++++++++++- 4 files changed, 375 insertions(+), 20 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 4dbaf5b00..1f71c05aa 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -7,6 +7,9 @@ ### Changed - Enforced `all_turns` reasoning context for all Responses Lite requests +### Fixed + +- Fixed xAI OAuth Responses continuations replaying OpenAI-only `custom_tool_call`/`custom_tool_call_output` history and `input_image.detail: "original"` frames; replay now downgrades those to xAI-compatible function calls and `detail: "auto"`. ([#5002](https://github.com/can1357/oh-my-pi/issues/5002)) ## [16.4.0] - 2026-07-10 diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index f70989560..afb15ba97 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1356,6 +1356,52 @@ export function convertResponsesInputContent( return normalizedContent.length > 0 ? normalizedContent : undefined; } +interface ResponsesReplayCompatibilityOptions { + supportsCustomToolCalls: boolean; + tools: readonly Tool[] | undefined; +} + +function resolveReplayCustomToolName(wireName: string, tools: readonly Tool[] | undefined): string { + if (tools) { + for (const tool of tools) { + if (tool.customWireName === wireName) return tool.name; + } + } + if (wireName === "apply_patch") return "edit"; + return wireName; +} + +function adaptResponsesReplayItemsForModel( + input: ResponseInput, + options: ResponsesReplayCompatibilityOptions, +): ResponseInput { + let changed = false; + const adapted: ResponseInput = []; + for (const item of input) { + let next = item; + if (!options.supportsCustomToolCalls && item.type === "custom_tool_call") { + changed = true; + next = { + type: "function_call", + ...(item.id ? { id: item.id } : {}), + call_id: item.call_id, + name: resolveReplayCustomToolName(item.name, options.tools), + arguments: JSON.stringify({ input: item.input }), + ...(item.namespace ? { namespace: item.namespace } : {}), + }; + } else if (!options.supportsCustomToolCalls && item.type === "custom_tool_call_output") { + changed = true; + next = { + type: "function_call_output", + call_id: item.call_id, + output: item.output, + }; + } + adapted.push(next); + } + return changed ? adapted : input; +} + export interface BuildResponsesInputOptions { model: Model; context: Context; @@ -1380,6 +1426,13 @@ export function buildResponsesInput(options: BuildResponsesInp messages.push({ role: options.systemRole as "system" | "developer", content: systemPrompt }); } + const supportsImageDetailOriginal = + options.model.provider === "xai-oauth" ? false : options.supportsImageDetailOriginal; + const supportsCustomToolCalls = options.model.applyPatchToolType === "freeform"; + const replayCompatibility: ResponsesReplayCompatibilityOptions = { + supportsCustomToolCalls, + tools: options.context.tools, + }; let knownCallIds = new Set(); const customCallIds = new Set(); const transformedMessages = transformMessages( @@ -1407,7 +1460,10 @@ export function buildResponsesInput(options: BuildResponsesInp }) ?? false); if (historyItems && shouldReplayPayloadItems) { - messages.push(...sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems))); + const sanitizedItems = sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems), { + supportsImageDetailOriginal, + }); + messages.push(...adaptResponsesReplayItemsForModel(sanitizedItems, replayCompatibility)); knownCallIds = collectKnownCallIds(messages); for (const id of collectCustomCallIds(messages)) customCallIds.add(id); msgIndex++; @@ -1416,7 +1472,7 @@ export function buildResponsesInput(options: BuildResponsesInp const content = convertResponsesInputContent( msg.content, options.model.input.includes("image"), - options.supportsImageDetailOriginal, + supportsImageDetailOriginal, ); if (!content) continue; messages.push({ @@ -1444,9 +1500,13 @@ export function buildResponsesInput(options: BuildResponsesInp const historyItems = providerPayload?.items; let suppressHiddenEmptyFallback = false; if (historyItems) { - const sanitizedHistoryItems = sanitizeOpenAIResponsesAssistantHistoryItemsForReplay( + const rawSanitizedHistoryItems = sanitizeOpenAIResponsesAssistantHistoryItemsForReplay( filterReasoning(historyItems), + { supportsImageDetailOriginal }, ); + const sanitizedHistoryItems = rawSanitizedHistoryItems + ? adaptResponsesReplayItemsForModel(rawSanitizedHistoryItems, replayCompatibility) + : undefined; if (nativeReplayEnabled && sanitizedHistoryItems) { if (providerPayload?.dt) { messages.push(...sanitizedHistoryItems); @@ -1469,6 +1529,8 @@ export function buildResponsesInput(options: BuildResponsesInp suppressHiddenEmptyFallback ? false : includeThinkingSignatures, customCallIds, options.preserveAssistantMessageIds, + supportsCustomToolCalls, + options.context.tools, ); const outputItems = suppressHiddenEmptyFallback ? sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(convertedOutputItems) @@ -1481,9 +1543,10 @@ export function buildResponsesInput(options: BuildResponsesInp msg, options.model, options.strictResponsesPairing, - options.supportsImageDetailOriginal, + supportsImageDetailOriginal, knownCallIds, customCallIds, + supportsCustomToolCalls, ); } msgIndex++; @@ -1516,6 +1579,8 @@ export function convertResponsesAssistantMessage( includeThinkingSignatures = true, customCallIds?: Set, preserveMessageIds = false, + supportsCustomToolCalls = true, + tools?: readonly Tool[], ): ResponseInput { const outputItems: ResponseInput = []; let unsignedTextBlocks = 0; @@ -1587,7 +1652,7 @@ export function convertResponsesAssistantMessage( itemId = undefined; } knownCallIds.add(normalized.callId); - if (block.customWireName) { + if (block.customWireName && supportsCustomToolCalls) { const rawInput = typeof block.arguments?.input === "string" ? block.arguments.input : ""; customCallIds?.add(normalized.callId); outputItems.push({ @@ -1599,11 +1664,15 @@ export function convertResponsesAssistantMessage( } as ResponseInput[number]); continue; } + const functionName = + block.customWireName && !supportsCustomToolCalls + ? resolveReplayCustomToolName(block.customWireName, tools) + : block.name; outputItems.push({ type: "function_call", ...(itemId ? { id: itemId } : {}), call_id: normalized.callId, - name: block.name, + name: functionName, arguments: JSON.stringify(block.arguments), }); } @@ -1619,6 +1688,7 @@ export function appendResponsesToolResultMessages( supportsImageDetailOriginal: boolean, knownCallIds: ReadonlySet, customCallIds?: ReadonlySet, + supportsCustomToolCalls = true, ): void { const supportsImages = model.input.includes("image"); const textResult = toolResult.content @@ -1648,7 +1718,7 @@ export function appendResponsesToolResultMessages( } as ResponseInput[number]); return; } - if (customCallIds?.has(normalized.callId)) { + if (supportsCustomToolCalls && customCallIds?.has(normalized.callId)) { messages.push({ type: "custom_tool_call_output", call_id: normalized.callId, diff --git a/packages/ai/src/utils.ts b/packages/ai/src/utils.ts index 0abc70cb1..988c16aa5 100644 --- a/packages/ai/src/utils.ts +++ b/packages/ai/src/utils.ts @@ -65,10 +65,50 @@ export function truncateResponseItemId(id: string, prefix: string): string { return `${prefix}_${Bun.hash(id).toString(36)}`; } -export function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Array>): ResponseInput { +interface OpenAIResponsesReplaySanitizeOptions { + supportsImageDetailOriginal?: boolean; +} + +function isReplayRecord(value: unknown): value is Record { + if (!value || typeof value !== "object") return false; + return !Array.isArray(value); +} + +function sanitizeReplayValueForCompatibility(value: unknown, options: OpenAIResponsesReplaySanitizeOptions): unknown { + if (options.supportsImageDetailOriginal !== false) return value; + if (Array.isArray(value)) { + let changed = false; + const sanitized = value.map(item => { + const next = sanitizeReplayValueForCompatibility(item, options); + if (next !== item) changed = true; + return next; + }); + return changed ? sanitized : value; + } + if (!isReplayRecord(value)) return value; + + let changed = false; + const sanitized: Record = {}; + for (const key in value) { + const child = value[key]; + const next = sanitizeReplayValueForCompatibility(child, options); + if (next !== child) changed = true; + sanitized[key] = next; + } + if (value.type === "input_image" && value.detail === "original") { + sanitized.detail = "auto"; + changed = true; + } + return changed ? sanitized : value; +} + +export function sanitizeOpenAIResponsesHistoryItemsForReplay( + items: Array>, + options: OpenAIResponsesReplaySanitizeOptions = {}, +): ResponseInput { const normalizedCallIds = new Map(); return items.flatMap(item => { - const sanitized = sanitizeOpenAIResponsesHistoryItemForReplay(item, normalizedCallIds); + const sanitized = sanitizeOpenAIResponsesHistoryItemForReplay(item, normalizedCallIds, options); return sanitized ? [sanitized] : []; }); } @@ -82,8 +122,9 @@ export function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Array>, + options: OpenAIResponsesReplaySanitizeOptions = {}, ): ResponseInput | undefined { - const sanitized = sanitizeOpenAIResponsesHistoryItemsForReplay(items); + const sanitized = sanitizeOpenAIResponsesHistoryItemsForReplay(items, options); let hasReplayableAssistantOutput = false; for (const item of sanitized) { @@ -153,6 +194,7 @@ export function sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(items: Re function sanitizeOpenAIResponsesHistoryItemForReplay( item: Record, normalizedCallIds: Map, + options: OpenAIResponsesReplaySanitizeOptions, ): OpenAIResponsesReplayItem | undefined { if (item.type === "item_reference") return undefined; if (item.type === "image_generation_call") return sanitizeOpenAIResponsesImageGenerationCallForReplay(item); @@ -164,7 +206,8 @@ function sanitizeOpenAIResponsesHistoryItemForReplay( sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds); } - return sanitizedItem as unknown as OpenAIResponsesReplayItem; + const compatibleItem = sanitizeReplayValueForCompatibility(sanitizedItem, options); + return compatibleItem as unknown as OpenAIResponsesReplayItem; } function sanitizeOpenAIResponsesReasoningItemForReplay(item: Record): OpenAIResponsesReplayItem { diff --git a/packages/ai/test/openai-responses-history-payload.test.ts b/packages/ai/test/openai-responses-history-payload.test.ts index 9520fabf0..a1a4dc84e 100644 --- a/packages/ai/test/openai-responses-history-payload.test.ts +++ b/packages/ai/test/openai-responses-history-payload.test.ts @@ -5,11 +5,12 @@ import { } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { type OpenAIResponsesOptions, streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import { buildResponsesInput } from "@oh-my-pi/pi-ai/providers/openai-shared"; -import type { Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import type { Context, Model, ModelSpec, ProviderSessionState, Tool } from "@oh-my-pi/pi-ai/types"; import { createOpenAIResponsesHistoryPayload, truncateResponseItemId } from "@oh-my-pi/pi-ai/utils"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { type GeneratedProvider, getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as piUtils from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; @@ -35,13 +36,42 @@ function createCodexToken(accountId: string): string { return `${header}.${payload}.signature`; } -function getOpenAIReasoningModel( - provider: Parameters[0], - id: string, -): Model<"openai-responses"> { - return getBundledModel(provider, id) as Model<"openai-responses">; +function getOpenAIReasoningModel(provider: GeneratedProvider, id: string): Model<"openai-responses"> { + const model = getBundledModel<"openai-responses">(provider, id); + return model; } +const ISSUE_5002_PATCH = "*** Begin Patch\n*** End Patch\n"; +const ISSUE_5002_TOOL_OUTPUT = "patch applied"; +const issue5002XaiOAuthModel = buildModel({ + id: "grok-build", + name: "Grok Build", + api: "openai-responses", + provider: "xai-oauth", + baseUrl: "https://api.x.ai/v1", + reasoning: true, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 256000, + maxTokens: 64000, +} satisfies ModelSpec<"openai-responses">); + +const issue5002ZeroUsage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; +const issue5002EditTool: Tool = { + name: "edit", + customWireName: "apply_patch", + description: "Apply a hashline patch", + parameters: type({ input: "string" }), + customFormat: { syntax: "lark", definition: 'start: "*** Begin Patch" LF\nLF: /\\n/' }, +}; + const preservedHistoryItems = [ { type: "message", role: "user", content: [{ type: "input_text", text: "Preserved user" }] }, { type: "compaction", encrypted_content: "enc_123" }, @@ -298,6 +328,38 @@ function findResponsesInputItem(input: unknown[] | undefined, type: string): Rec }) as Record | undefined; } +function isIssue5002Record(value: unknown): value is Record { + if (value === null || typeof value !== "object" || Array.isArray(value)) return false; + return true; +} + +function findResponsesInputItemByCallId( + input: unknown[], + type: string, + callId: string, +): Record | undefined { + for (const item of input) { + if (!isIssue5002Record(item)) continue; + if (item.type === type && item.call_id === callId) return item; + } + return undefined; +} + +function collectResponsesInputImageDetails(input: unknown): string[] { + const details: string[] = []; + const visit = (node: unknown): void => { + if (Array.isArray(node)) { + for (const child of node) visit(child); + return; + } + if (!isIssue5002Record(node)) return; + if (node.type === "input_image" && typeof node.detail === "string") details.push(node.detail); + for (const key in node) visit(node[key]); + }; + visit(input); + return details; +} + function containsUserInputText(input: unknown[] | undefined, text: string): boolean { return (input ?? []).some(item => { if (!item || typeof item !== "object") return false; @@ -366,11 +428,188 @@ describe("OpenAI responses history payload", () => { }); assertWireOrder(openaiItems); - const codexModel = getBundledModel("openai-codex", "gpt-5.2-codex") as Model<"openai-codex-responses">; + const codexModel = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.2-codex"); const codexItems = convertCodexResponsesMessages(codexModel, makeContext("openai-codex")); assertWireOrder(codexItems); }); + it("adapts reconstructed apply_patch replay for xai-oauth while preserving OpenAI custom replay", () => { + const context: Context = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "previous frame" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==", detail: "original" }, + ], + timestamp: Date.now(), + }, + { + role: "assistant", + content: [ + { + type: "toolCall", + id: "call_apply", + name: "apply_patch", + arguments: { input: ISSUE_5002_PATCH }, + customWireName: "apply_patch", + }, + ], + api: "openai-responses", + provider: "openai", + model: "gpt-5-mini", + usage: issue5002ZeroUsage, + stopReason: "toolUse", + timestamp: Date.now(), + }, + { + role: "toolResult", + toolCallId: "call_apply", + toolName: "edit", + content: [{ type: "text", text: ISSUE_5002_TOOL_OUTPUT }], + isError: false, + timestamp: Date.now(), + }, + ], + tools: [issue5002EditTool], + }; + + const xaiInput = buildResponsesInput({ + model: issue5002XaiOAuthModel, + context, + strictResponsesPairing: false, + supportsImageDetailOriginal: issue5002XaiOAuthModel.compat.supportsImageDetailOriginal, + nativeHistory: { replay: true, filterReasoning: issue5002XaiOAuthModel.compat.filterReasoningHistory }, + }); + expect(findResponsesInputItemByCallId(xaiInput, "function_call", "call_apply")).toEqual({ + type: "function_call", + call_id: "call_apply", + name: "edit", + arguments: JSON.stringify({ input: ISSUE_5002_PATCH }), + }); + expect(findResponsesInputItemByCallId(xaiInput, "function_call_output", "call_apply")).toEqual({ + type: "function_call_output", + call_id: "call_apply", + output: ISSUE_5002_TOOL_OUTPUT, + }); + expect(JSON.stringify(xaiInput)).not.toContain("custom_tool_call"); + expect(collectResponsesInputImageDetails(xaiInput)).toEqual(["auto"]); + + const openaiModel = getOpenAIReasoningModel("openai", "gpt-5-mini"); + const openaiInput = buildResponsesInput({ + model: openaiModel, + context, + strictResponsesPairing: false, + supportsImageDetailOriginal: openaiModel.compat.supportsImageDetailOriginal, + nativeHistory: { replay: true, filterReasoning: openaiModel.compat.filterReasoningHistory }, + }); + expect(findResponsesInputItemByCallId(openaiInput, "custom_tool_call", "call_apply")).toEqual({ + type: "custom_tool_call", + call_id: "call_apply", + name: "apply_patch", + input: ISSUE_5002_PATCH, + }); + expect(findResponsesInputItemByCallId(openaiInput, "custom_tool_call_output", "call_apply")).toEqual({ + type: "custom_tool_call_output", + call_id: "call_apply", + output: ISSUE_5002_TOOL_OUTPUT, + }); + expect(collectResponsesInputImageDetails(openaiInput)).toEqual(["original"]); + }); + + it("adapts persisted native apply_patch Responses items for xai-oauth continuations", () => { + const nativeHistoryItems = [ + { + type: "message", + role: "user", + content: [ + { type: "input_text", text: "previous native frame" }, + { type: "input_image", detail: "original", image_url: "data:image/png;base64,ZmFrZQ==" }, + ], + }, + { type: "custom_tool_call", call_id: "call_native_apply", name: "apply_patch", input: ISSUE_5002_PATCH }, + { + type: "custom_tool_call_output", + call_id: "call_native_apply", + output: ISSUE_5002_TOOL_OUTPUT, + }, + ]; + const xaiContext: Context = { + messages: [ + { + role: "assistant", + content: [{ type: "text", text: "fallback should not be replayed" }], + api: "openai-responses", + provider: "xai-oauth", + model: issue5002XaiOAuthModel.id, + usage: issue5002ZeroUsage, + stopReason: "stop", + providerPayload: createOpenAIResponsesHistoryPayload("xai-oauth", nativeHistoryItems), + timestamp: Date.now(), + }, + { role: "user", content: "continue", timestamp: Date.now() }, + ], + }; + + const xaiInput = buildResponsesInput({ + model: issue5002XaiOAuthModel, + context: xaiContext, + strictResponsesPairing: false, + supportsImageDetailOriginal: issue5002XaiOAuthModel.compat.supportsImageDetailOriginal, + nativeHistory: { replay: true, filterReasoning: issue5002XaiOAuthModel.compat.filterReasoningHistory }, + }); + expect(findResponsesInputItemByCallId(xaiInput, "function_call", "call_native_apply")).toEqual({ + type: "function_call", + call_id: "call_native_apply", + name: "edit", + arguments: JSON.stringify({ input: ISSUE_5002_PATCH }), + }); + expect(findResponsesInputItemByCallId(xaiInput, "function_call_output", "call_native_apply")).toEqual({ + type: "function_call_output", + call_id: "call_native_apply", + output: ISSUE_5002_TOOL_OUTPUT, + }); + expect(JSON.stringify(xaiInput)).not.toContain("custom_tool_call"); + expect(collectResponsesInputImageDetails(xaiInput)).toEqual(["auto"]); + + const openaiModel = getOpenAIReasoningModel("openai", "gpt-5-mini"); + const openaiContext: Context = { + messages: [ + { + role: "assistant", + content: [{ type: "text", text: "fallback should not be replayed" }], + api: "openai-responses", + provider: "openai", + model: openaiModel.id, + usage: issue5002ZeroUsage, + stopReason: "stop", + providerPayload: createOpenAIResponsesHistoryPayload("openai", nativeHistoryItems), + timestamp: Date.now(), + }, + { role: "user", content: "continue", timestamp: Date.now() }, + ], + }; + const openaiInput = buildResponsesInput({ + model: openaiModel, + context: openaiContext, + strictResponsesPairing: false, + supportsImageDetailOriginal: openaiModel.compat.supportsImageDetailOriginal, + nativeHistory: { replay: true, filterReasoning: openaiModel.compat.filterReasoningHistory }, + }); + expect(findResponsesInputItemByCallId(openaiInput, "custom_tool_call", "call_native_apply")).toEqual({ + type: "custom_tool_call", + call_id: "call_native_apply", + name: "apply_patch", + input: ISSUE_5002_PATCH, + }); + expect(findResponsesInputItemByCallId(openaiInput, "custom_tool_call_output", "call_native_apply")).toEqual({ + type: "custom_tool_call_output", + call_id: "call_native_apply", + output: ISSUE_5002_TOOL_OUTPUT, + }); + expect(collectResponsesInputImageDetails(openaiInput)).toEqual(["original"]); + }); + it("prepends multiple OpenAI developer instructions in order without changing prompt cache key routing", async () => { const model = getOpenAIReasoningModel("openai", "gpt-5-mini"); const payload = (await captureResponsesPayload( @@ -1063,7 +1302,7 @@ describe("OpenAI responses history payload", () => { { role: "user", content: "Resume", timestamp: Date.now() }, ], }; - const model = getBundledModel("openai-codex", "gpt-5.2-codex") as Model<"openai-codex-responses">; + const model = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.2-codex"); const payload = (await captureCodexPayload(model, context)) as { input?: unknown[] }; const functionCallItem = findResponsesInputItem(payload.input, "function_call"); const functionCallOutputItem = findResponsesInputItem(payload.input, "function_call_output"); From 619519343595a10bca67fac076f09c41c37caa80 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 15:24:27 -0300 Subject: [PATCH 090/205] fix(catalog): marked xai-oauth models without image detail original Seed supportsImageDetailOriginal=false for curated and dynamic xai-oauth compat, and set the same flag on the eight bundled models.json entries without regenerating the rest of the catalog. Fixes #5002 --- packages/catalog/src/models.json | 24 ++++++++++++------- .../src/provider-models/openai-compat.ts | 2 ++ 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 80bb13350..43840585a 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -87250,7 +87250,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": true, - "supportsReasoningEffort": false + "supportsReasoningEffort": false, + "supportsImageDetailOriginal": false } }, "grok-4.20-0309-reasoning": { @@ -87279,7 +87280,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": true, - "supportsReasoningEffort": false + "supportsReasoningEffort": false, + "supportsImageDetailOriginal": false } }, "grok-4.20-multi-agent-0309": { @@ -87320,7 +87322,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": false, - "supportsReasoningEffort": true + "supportsReasoningEffort": true, + "supportsImageDetailOriginal": false } }, "grok-4.3": { @@ -87362,7 +87365,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": false, - "supportsReasoningEffort": true + "supportsReasoningEffort": true, + "supportsImageDetailOriginal": false } }, "grok-4.5": { @@ -87404,7 +87408,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": false, - "supportsReasoningEffort": true + "supportsReasoningEffort": true, + "supportsImageDetailOriginal": false } }, "grok-build": { @@ -87433,7 +87438,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": true, - "supportsReasoningEffort": false + "supportsReasoningEffort": false, + "supportsImageDetailOriginal": false } }, "grok-build-0.1": { @@ -87462,7 +87468,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": true, - "supportsReasoningEffort": false + "supportsReasoningEffort": false, + "supportsImageDetailOriginal": false } }, "grok-composer-2.5-fast": { @@ -87490,7 +87497,8 @@ "includeEncryptedReasoning": false, "filterReasoningHistory": true, "omitReasoningEffort": true, - "supportsReasoningEffort": false + "supportsReasoningEffort": false, + "supportsImageDetailOriginal": false } } }, diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index ef7f7eafa..40ab74cd8 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1176,6 +1176,7 @@ function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): Model ...(model.compat ?? {}), includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false, filterReasoningHistory: model.compat?.filterReasoningHistory ?? true, + supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false, omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id), }; return { ...model, compat }; @@ -1218,6 +1219,7 @@ function mergeCuratedIntoModel( reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) }, includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false, filterReasoningHistory: base.compat?.filterReasoningHistory ?? true, + supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false, omitReasoningEffort: !effortCapable, supportsReasoningEffort: effortCapable, }; From 0e3e5ab72a5f5ebac1d03339ca2707bc005bd617 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 15:27:24 -0300 Subject: [PATCH 091/205] fix(ai): tightened xai replay adaptation to catalog compat Drive image-detail clamping from resolved model compat instead of a provider hardcode, clamp input_image only on known paths, and skip custom-tool adaptation when freeform is supported. --- packages/ai/src/providers/openai-shared.ts | 87 ++++++++++++++-------- packages/ai/src/utils.ts | 59 ++++++++------- packages/catalog/src/compat/openai.ts | 14 ++-- 3 files changed, 91 insertions(+), 69 deletions(-) diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index afb15ba97..ca6e67771 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1356,48 +1356,61 @@ export function convertResponsesInputContent( return normalizedContent.length > 0 ? normalizedContent : undefined; } -interface ResponsesReplayCompatibilityOptions { - supportsCustomToolCalls: boolean; - tools: readonly Tool[] | undefined; -} - -function resolveReplayCustomToolName(wireName: string, tools: readonly Tool[] | undefined): string { - if (tools) { - for (const tool of tools) { - if (tool.customWireName === wireName) return tool.name; - } +/** + * Map freeform custom-tool wire names back to the internal tool name for + * providers that only accept function_call / function_call_output. + * Built once per request; `apply_patch` → `edit` is the OMP default. + */ +function buildCustomToolWireNameMap(tools: readonly Tool[] | undefined): ReadonlyMap | undefined { + if (!tools?.length) return undefined; + const map = new Map(); + for (const tool of tools) { + if (tool.customWireName) map.set(tool.customWireName, tool.name); } - if (wireName === "apply_patch") return "edit"; - return wireName; + return map.size > 0 ? map : undefined; } +function resolveReplayCustomToolName(wireName: string, wireNameMap: ReadonlyMap | undefined): string { + return wireNameMap?.get(wireName) ?? (wireName === "apply_patch" ? "edit" : wireName); +} + +/** + * Downgrade OpenAI-only custom tool items when the target model does not + * advertise freeform custom tools (`applyPatchToolType === "freeform"`). + * No-op (returns the same array reference) when freeform is supported. + */ function adaptResponsesReplayItemsForModel( input: ResponseInput, - options: ResponsesReplayCompatibilityOptions, + supportsCustomToolCalls: boolean, + wireNameMap: ReadonlyMap | undefined, ): ResponseInput { + if (supportsCustomToolCalls) return input; + let changed = false; const adapted: ResponseInput = []; for (const item of input) { - let next = item; - if (!options.supportsCustomToolCalls && item.type === "custom_tool_call") { + if (item.type === "custom_tool_call") { changed = true; - next = { + adapted.push({ type: "function_call", ...(item.id ? { id: item.id } : {}), call_id: item.call_id, - name: resolveReplayCustomToolName(item.name, options.tools), + name: resolveReplayCustomToolName(item.name, wireNameMap), arguments: JSON.stringify({ input: item.input }), ...(item.namespace ? { namespace: item.namespace } : {}), - }; - } else if (!options.supportsCustomToolCalls && item.type === "custom_tool_call_output") { + }); + continue; + } + if (item.type === "custom_tool_call_output") { changed = true; - next = { + adapted.push({ type: "function_call_output", call_id: item.call_id, output: item.output, - }; + }); + continue; } - adapted.push(next); + adapted.push(item); } return changed ? adapted : input; } @@ -1426,13 +1439,15 @@ export function buildResponsesInput(options: BuildResponsesInp messages.push({ role: options.systemRole as "system" | "developer", content: systemPrompt }); } - const supportsImageDetailOriginal = - options.model.provider === "xai-oauth" ? false : options.supportsImageDetailOriginal; + // Compat is resolved by the catalog (e.g. Copilot / xai-oauth reject + // `detail: "original"`). Do not re-branch on provider id here. + const supportsImageDetailOriginal = options.supportsImageDetailOriginal; + // Freeform custom tools (`custom_tool_call`) only when the catalog says so; + // same gate as tool conversion (`applyPatchToolType === "freeform"`). const supportsCustomToolCalls = options.model.applyPatchToolType === "freeform"; - const replayCompatibility: ResponsesReplayCompatibilityOptions = { - supportsCustomToolCalls, - tools: options.context.tools, - }; + const customToolWireNameMap = supportsCustomToolCalls + ? undefined + : buildCustomToolWireNameMap(options.context.tools); let knownCallIds = new Set(); const customCallIds = new Set(); const transformedMessages = transformMessages( @@ -1463,7 +1478,9 @@ export function buildResponsesInput(options: BuildResponsesInp const sanitizedItems = sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems), { supportsImageDetailOriginal, }); - messages.push(...adaptResponsesReplayItemsForModel(sanitizedItems, replayCompatibility)); + messages.push( + ...adaptResponsesReplayItemsForModel(sanitizedItems, supportsCustomToolCalls, customToolWireNameMap), + ); knownCallIds = collectKnownCallIds(messages); for (const id of collectCustomCallIds(messages)) customCallIds.add(id); msgIndex++; @@ -1505,7 +1522,11 @@ export function buildResponsesInput(options: BuildResponsesInp { supportsImageDetailOriginal }, ); const sanitizedHistoryItems = rawSanitizedHistoryItems - ? adaptResponsesReplayItemsForModel(rawSanitizedHistoryItems, replayCompatibility) + ? adaptResponsesReplayItemsForModel( + rawSanitizedHistoryItems, + supportsCustomToolCalls, + customToolWireNameMap, + ) : undefined; if (nativeReplayEnabled && sanitizedHistoryItems) { if (providerPayload?.dt) { @@ -1530,7 +1551,7 @@ export function buildResponsesInput(options: BuildResponsesInp customCallIds, options.preserveAssistantMessageIds, supportsCustomToolCalls, - options.context.tools, + customToolWireNameMap, ); const outputItems = suppressHiddenEmptyFallback ? sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(convertedOutputItems) @@ -1580,7 +1601,7 @@ export function convertResponsesAssistantMessage( customCallIds?: Set, preserveMessageIds = false, supportsCustomToolCalls = true, - tools?: readonly Tool[], + customToolWireNameMap?: ReadonlyMap, ): ResponseInput { const outputItems: ResponseInput = []; let unsignedTextBlocks = 0; @@ -1666,7 +1687,7 @@ export function convertResponsesAssistantMessage( } const functionName = block.customWireName && !supportsCustomToolCalls - ? resolveReplayCustomToolName(block.customWireName, tools) + ? resolveReplayCustomToolName(block.customWireName, customToolWireNameMap) : block.name; outputItems.push({ type: "function_call", diff --git a/packages/ai/src/utils.ts b/packages/ai/src/utils.ts index 988c16aa5..6e9cc5885 100644 --- a/packages/ai/src/utils.ts +++ b/packages/ai/src/utils.ts @@ -69,37 +69,32 @@ interface OpenAIResponsesReplaySanitizeOptions { supportsImageDetailOriginal?: boolean; } -function isReplayRecord(value: unknown): value is Record { - if (!value || typeof value !== "object") return false; - return !Array.isArray(value); -} +/** + * Clamp `detail: "original"` only where Responses input_image parts live — + * top-level items and `message.content[]`. Avoids a deep tree walk/clone of + * every history node on providers that reject native-resolution images. + */ +function clampReplayItemImageDetail( + item: Record, + supportsImageDetailOriginal: boolean, +): Record { + if (supportsImageDetailOriginal) return item; -function sanitizeReplayValueForCompatibility(value: unknown, options: OpenAIResponsesReplaySanitizeOptions): unknown { - if (options.supportsImageDetailOriginal !== false) return value; - if (Array.isArray(value)) { - let changed = false; - const sanitized = value.map(item => { - const next = sanitizeReplayValueForCompatibility(item, options); - if (next !== item) changed = true; - return next; - }); - return changed ? sanitized : value; + if (item.type === "input_image" && item.detail === "original") { + return { ...item, detail: "auto" }; } - if (!isReplayRecord(value)) return value; + + if (item.type !== "message" || !Array.isArray(item.content)) return item; let changed = false; - const sanitized: Record = {}; - for (const key in value) { - const child = value[key]; - const next = sanitizeReplayValueForCompatibility(child, options); - if (next !== child) changed = true; - sanitized[key] = next; - } - if (value.type === "input_image" && value.detail === "original") { - sanitized.detail = "auto"; + const content = item.content.map(part => { + if (!part || typeof part !== "object" || Array.isArray(part)) return part; + const record = part as Record; + if (record.type !== "input_image" || record.detail !== "original") return part; changed = true; - } - return changed ? sanitized : value; + return { ...record, detail: "auto" }; + }); + return changed ? { ...item, content } : item; } export function sanitizeOpenAIResponsesHistoryItemsForReplay( @@ -107,8 +102,13 @@ export function sanitizeOpenAIResponsesHistoryItemsForReplay( options: OpenAIResponsesReplaySanitizeOptions = {}, ): ResponseInput { const normalizedCallIds = new Map(); + const supportsImageDetailOriginal = options.supportsImageDetailOriginal !== false; return items.flatMap(item => { - const sanitized = sanitizeOpenAIResponsesHistoryItemForReplay(item, normalizedCallIds, options); + const sanitized = sanitizeOpenAIResponsesHistoryItemForReplay( + item, + normalizedCallIds, + supportsImageDetailOriginal, + ); return sanitized ? [sanitized] : []; }); } @@ -194,7 +194,7 @@ export function sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(items: Re function sanitizeOpenAIResponsesHistoryItemForReplay( item: Record, normalizedCallIds: Map, - options: OpenAIResponsesReplaySanitizeOptions, + supportsImageDetailOriginal: boolean, ): OpenAIResponsesReplayItem | undefined { if (item.type === "item_reference") return undefined; if (item.type === "image_generation_call") return sanitizeOpenAIResponsesImageGenerationCallForReplay(item); @@ -206,8 +206,7 @@ function sanitizeOpenAIResponsesHistoryItemForReplay( sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds); } - const compatibleItem = sanitizeReplayValueForCompatibility(sanitizedItem, options); - return compatibleItem as unknown as OpenAIResponsesReplayItem; + return clampReplayItemImageDetail(sanitizedItem, supportsImageDetailOriginal) as unknown as OpenAIResponsesReplayItem; } function sanitizeOpenAIResponsesReasoningItemForReplay(item: Record): OpenAIResponsesReplayItem { diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index a01430b33..1ced5e25c 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -595,12 +595,14 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol // Azure OpenAI and GitHub Copilot Responses paths require tool results // to strictly match prior tool calls when building Responses inputs. strictResponsesPairing: isAzure || spec.provider === "github-copilot", - // GitHub Copilot's Responses endpoint rejects the `detail: "original"` - // image hint with a 400; every other host preserves native-resolution - // frames (snapcompact relies on `original`). Detect Copilot by provider id - // or base-URL host (mirroring the Anthropic compat builder) so a model - // pointed at the Copilot host under a different provider id still clamps. - supportsImageDetailOriginal: !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"), + // GitHub Copilot and xAI OAuth reject `detail: "original"` (400 / 422). + // Every other host preserves native-resolution frames (snapcompact relies + // on `original`). Detect Copilot by provider id or base-URL host so a + // model pointed at the Copilot host under a different provider id still + // clamps; xai-oauth is provider-id only (same host family as paid `xai`). + supportsImageDetailOriginal: + spec.provider !== "xai-oauth" && + !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"), reasoningEffortMap: {}, supportsReasoningParams: true, thinkingFormat, From bd216f4394c5edbf3e6336429c228032b32063fc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 15:29:36 -0300 Subject: [PATCH 092/205] style(ai): applied biome format to xai replay changes --- packages/ai/src/utils.ts | 5 ++++- packages/catalog/src/compat/openai.ts | 3 +-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/packages/ai/src/utils.ts b/packages/ai/src/utils.ts index 6e9cc5885..0445cd3ac 100644 --- a/packages/ai/src/utils.ts +++ b/packages/ai/src/utils.ts @@ -206,7 +206,10 @@ function sanitizeOpenAIResponsesHistoryItemForReplay( sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds); } - return clampReplayItemImageDetail(sanitizedItem, supportsImageDetailOriginal) as unknown as OpenAIResponsesReplayItem; + return clampReplayItemImageDetail( + sanitizedItem, + supportsImageDetailOriginal, + ) as unknown as OpenAIResponsesReplayItem; } function sanitizeOpenAIResponsesReasoningItemForReplay(item: Record): OpenAIResponsesReplayItem { diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 1ced5e25c..2425a4744 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -601,8 +601,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol // model pointed at the Copilot host under a different provider id still // clamps; xai-oauth is provider-id only (same host family as paid `xai`). supportsImageDetailOriginal: - spec.provider !== "xai-oauth" && - !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"), + spec.provider !== "xai-oauth" && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"), reasoningEffortMap: {}, supportsReasoningParams: true, thinkingFormat, From e22ee0b6688b26f8453d54aef98d10f238fa4cac Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 16:26:19 -0300 Subject: [PATCH 093/205] docs(ai): moved xAI replay fix changelog to Unreleased MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebase onto 16.4.1 landed the entry under the released section; keep release intent under Unreleased. Dropped the broad models.json regen that conflicted — surgical xai-oauth-only catalog edits remain. --- packages/ai/CHANGELOG.md | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 1f71c05aa..000fda5c4 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,14 +2,15 @@ ## [Unreleased] +### Fixed + +- Fixed xAI OAuth Responses continuations replaying OpenAI-only `custom_tool_call`/`custom_tool_call_output` history and `input_image.detail: "original"` frames; replay now downgrades those to xAI-compatible function calls and `detail: "auto"`. ([#5002](https://github.com/can1357/oh-my-pi/issues/5002)) + ## [16.4.1] - 2026-07-10 ### Changed - Enforced `all_turns` reasoning context for all Responses Lite requests -### Fixed - -- Fixed xAI OAuth Responses continuations replaying OpenAI-only `custom_tool_call`/`custom_tool_call_output` history and `input_image.detail: "original"` frames; replay now downgrades those to xAI-compatible function calls and `detail: "auto"`. ([#5002](https://github.com/can1357/oh-my-pi/issues/5002)) ## [16.4.0] - 2026-07-10 From 2b46d6711b33fdef3d5e379f04476d1803998631 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Victor=20Ara=C3=BAjo?= Date: Fri, 10 Jul 2026 16:28:40 -0300 Subject: [PATCH 094/205] chore: re-trigger CI after unrelated workspace-fast flake Prior run failed streamPiNative under parallel load; xAI replay and scoped models.json changes are orthogonal. Empty commit to re-run checks. From cf021ad39335fc7b9d99fb7a98119ade8374f9ea Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 20:25:49 +0000 Subject: [PATCH 095/205] fix(auth): serialized mcp oauth refreshes - Added durable SQLite refresh ownership for stored OAuth rows, with canonical re-read before refresh and compare-and-set persistence. - Routed MCP proactive and forced OAuth refresh through the shared owner so waiters reuse the winner's rotated credential. - Added MCP regression tests for shared SQLite refresh ownership and stale invalid_grant losers. Fixes #5081 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/auth-storage.ts | 284 +++++++++++++++++- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/mcp/manager.ts | 130 ++++---- .../test/mcp-manager-oauth-refresh.test.ts | 190 ++++++++++++ 5 files changed, 547 insertions(+), 65 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index eda640539..544ac2467 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed shared SQLite OAuth refreshes to use durable credential-row ownership plus compare-and-set persistence, preventing stale refresh failures from deleting or overwriting a peer's rotated credential. ([#5081](https://github.com/can1357/oh-my-pi/issues/5081)) + ## [16.4.0] - 2026-07-10 ### Added diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 417d1e1c8..20b777eea 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -314,6 +314,7 @@ export interface AuthCredentialStore { updateAuthCredential(id: number, credential: AuthCredential): void; deleteAuthCredential(id: number, disabledCause: string): void; tryDisableAuthCredentialIfMatches(id: number, expectedData: string, disabledCause: string): boolean; + tryUpdateAuthCredentialIfMatches?(id: number, expectedData: string, credential: AuthCredential): boolean; replaceAuthCredentialsForProvider(provider: string, credentials: AuthCredential[]): StoredAuthCredential[]; upsertAuthCredentialForProvider(provider: string, credential: AuthCredential): StoredAuthCredential[]; deleteAuthCredentialsForProvider(provider: string, disabledCause: string): void; @@ -332,6 +333,9 @@ export interface AuthCredentialStore { cleanExpiredCredentialBlocks?(nowMs: number): void; /** List non-expired blocks for broker snapshots. */ listCredentialBlocks?(credentialIds: readonly number[]): StoredCredentialBlock[]; + tryAcquireCredentialRefreshLease?(credentialId: number, owner: string, expiresAtMs: number): boolean; + getCredentialRefreshLeaseExpiresAt?(credentialId: number): number | undefined; + releaseCredentialRefreshLease?(credentialId: number, owner: string): void; /** * Append usage-limit snapshots for trend history. Optional: stores without * durable storage (e.g. the broker remote store) omit it and recording is @@ -593,6 +597,8 @@ const DEFAULT_OAUTH_REFRESH_TIMEOUT_MS = 10_000; * the rotation cadence by <4%. */ const OAUTH_REFRESH_SKEW_MS = 60_000; +const OAUTH_REFRESH_LEASE_TTL_MS = 15_000; +const OAUTH_REFRESH_LEASE_POLL_MS = 50; /** * Cap on the buffered credential_disabled backlog held while no handler is attached. * In practice the backlog is 0–N where N ≈ active providers (≤ ~20). The cap exists so @@ -718,6 +724,29 @@ export interface InvalidateCredentialMatchingOptions { sessionId?: string; } +/** Options for refreshing one stored OAuth row through durable ownership. */ +export interface StoredOAuthRefreshOptions { + observedCredential?: T; + credentialFromRow: (credential: OAuthCredential) => T | undefined; + forceRefresh?: boolean; + canRefresh?: (credential: T) => boolean; + refreshSkewMs?: number; + signal?: AbortSignal; + keepCredentialOnRefreshFailure?: boolean | ((error: unknown) => boolean); + onRefreshFailure?: (error: unknown) => void; + refresh: (credential: T) => Promise; + mergeRefreshedCredential?: (credential: T, refreshed: OAuthCredentials) => T; + isDefinitiveFailure?: (error: unknown) => boolean; + disabledCause?: (error: unknown) => string; +} + +/** Result of a stored OAuth refresh attempt. */ +export interface StoredOAuthRefreshResult { + credential: T | undefined; + refreshed: boolean; + removed: boolean; +} + /** * Identifies which stored account to redeem a saved rate-limit reset for. * Any one field is enough; `credentialId` is the most precise. @@ -1833,6 +1862,162 @@ export class AuthStorage { return rows; } + /** + * Refresh one stored OAuth credential under durable row ownership. + */ + async refreshStoredOAuthCredential( + provider: string, + options: StoredOAuthRefreshOptions, + ): Promise> { + const refreshSkewMs = options.refreshSkewMs ?? OAUTH_REFRESH_SKEW_MS; + const hasDurableLease = + !!this.#store.tryAcquireCredentialRefreshLease && + !!this.#store.getCredentialRefreshLeaseExpiresAt && + !!this.#store.releaseCredentialRefreshLease; + const owner = crypto.randomUUID(); + let leasedCredentialId: number | undefined; + + while (hasDurableLease) { + if (options.signal?.aborted) throw new AIError.AbortError("OAuth refresh ownership aborted by caller"); + const rows = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + rows.map(row => ({ id: row.id, credential: row.credential })), + ); + const row = rows.find(entry => entry.credential.type === "oauth"); + if (row?.credential.type !== "oauth") { + return { credential: undefined, refreshed: false, removed: false }; + } + const current = options.credentialFromRow(row.credential); + if (!current) { + return { credential: undefined, refreshed: false, removed: false }; + } + if (options.observedCredential && !authCredentialEquals(current, options.observedCredential)) { + return { credential: current, refreshed: false, removed: false }; + } + if (!options.forceRefresh && Date.now() + refreshSkewMs < current.expires) { + return { credential: current, refreshed: false, removed: false }; + } + if (options.canRefresh && !options.canRefresh(current)) { + return { credential: current, refreshed: false, removed: false }; + } + if (this.#store.tryAcquireCredentialRefreshLease?.(row.id, owner, Date.now() + OAUTH_REFRESH_LEASE_TTL_MS)) { + leasedCredentialId = row.id; + break; + } + const leaseExpiresAt = this.#store.getCredentialRefreshLeaseExpiresAt?.(row.id); + const waitMs = + leaseExpiresAt === undefined + ? OAUTH_REFRESH_LEASE_POLL_MS + : Math.min(Math.max(leaseExpiresAt - Date.now(), OAUTH_REFRESH_LEASE_POLL_MS), 250); + await raceCredentialRefreshWithSignal( + Bun.sleep(waitMs), + options.signal, + "OAuth refresh ownership wait aborted by caller", + ); + } + + try { + const rows = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + rows.map(row => ({ id: row.id, credential: row.credential })), + ); + const row = rows.find(entry => entry.credential.type === "oauth"); + if (row?.credential.type !== "oauth") { + return { credential: undefined, refreshed: false, removed: false }; + } + const current = options.credentialFromRow(row.credential); + if (!current) { + return { credential: undefined, refreshed: false, removed: false }; + } + if (options.observedCredential && !authCredentialEquals(current, options.observedCredential)) { + return { credential: current, refreshed: false, removed: false }; + } + if (!options.forceRefresh && Date.now() + refreshSkewMs < current.expires) { + return { credential: current, refreshed: false, removed: false }; + } + if (options.canRefresh && !options.canRefresh(current)) { + return { credential: current, refreshed: false, removed: false }; + } + const serialized = serializeCredential(provider, current); + if (!serialized) return { credential: current, refreshed: false, removed: false }; + + let refreshed: OAuthCredentials; + try { + refreshed = await options.refresh(current); + } catch (error) { + if (options.isDefinitiveFailure?.(error)) { + const disabledCause = options.disabledCause?.(error) ?? `oauth refresh failed: ${String(error)}`; + const disabled = this.#store.tryDisableAuthCredentialIfMatches(row.id, serialized.data, disabledCause); + if (disabled) { + this.#setStoredCredentials( + provider, + rows + .filter(entry => entry.id !== row.id) + .map(entry => ({ id: entry.id, credential: entry.credential })), + ); + this.#resetProviderAssignments(provider); + this.#emitCredentialDisabled({ provider, disabledCause }); + return { credential: undefined, refreshed: false, removed: true }; + } + await this.reload(); + const latest = this.get(provider); + return { + credential: latest?.type === "oauth" ? options.credentialFromRow(latest) : undefined, + refreshed: false, + removed: false, + }; + } + options.onRefreshFailure?.(error); + const keepCredential = + typeof options.keepCredentialOnRefreshFailure === "function" + ? options.keepCredentialOnRefreshFailure(error) + : options.keepCredentialOnRefreshFailure === true; + if (keepCredential) { + return { credential: current, refreshed: false, removed: false }; + } + throw error; + } + + const merged: T = options.mergeRefreshedCredential + ? options.mergeRefreshedCredential(current, refreshed) + : { + ...current, + access: refreshed.access, + refresh: refreshed.refresh, + expires: refreshed.expires, + accountId: refreshed.accountId ?? current.accountId, + email: refreshed.email ?? current.email, + projectId: refreshed.projectId ?? current.projectId, + enterpriseUrl: refreshed.enterpriseUrl ?? current.enterpriseUrl, + apiEndpoint: refreshed.apiEndpoint ?? current.apiEndpoint, + }; + if (this.#store.tryUpdateAuthCredentialIfMatches) { + if (!this.#store.tryUpdateAuthCredentialIfMatches(row.id, serialized.data, merged)) { + await this.reload(); + const latest = this.get(provider); + return { + credential: latest?.type === "oauth" ? options.credentialFromRow(latest) : undefined, + refreshed: false, + removed: false, + }; + } + } else { + this.#store.updateAuthCredential(row.id, merged); + } + this.#setStoredCredentials( + provider, + rows.map(entry => ({ id: entry.id, credential: entry.id === row.id ? merged : entry.credential })), + ); + return { credential: merged, refreshed: true, removed: false }; + } finally { + if (leasedCredentialId !== undefined) { + this.#store.releaseCredentialRefreshLease?.(leasedCredentialId, owner); + } + } + } + async #upsertOAuthCredential(provider: string, credential: OAuthCredential): Promise { const stored = this.#store.upsertAuthCredentialRemote ? await this.#store.upsertAuthCredentialRemote(provider, credential) @@ -5018,7 +5203,7 @@ type SerializedCredentialRecord = { identityKey: string | null; }; -const AUTH_SCHEMA_VERSION = 5; +const AUTH_SCHEMA_VERSION = 6; const SQLITE_NOW_EPOCH = "CAST(strftime('%s','now') AS INTEGER)"; /** @@ -5214,6 +5399,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { #updateStmt: Statement; #deleteStmt: Statement; #deleteIfMatchesStmt: Statement; + #updateIfMatchesStmt: Statement; #deleteByProviderStmt: Statement; #hardDeleteStmt: Statement; #getCacheStmt: Statement; @@ -5225,6 +5411,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { #upsertCredentialBlockStmt: Statement; #deleteCredentialBlocksStmt: Statement; #deleteExpiredCredentialBlocksStmt: Statement; + #acquireCredentialRefreshLeaseStmt: Statement; + #getCredentialRefreshLeaseStmt: Statement; + #releaseCredentialRefreshLeaseStmt: Statement; #credentialBlockReconcileAfter: Map = new Map(); #insertUsageHistoryStmt: Statement; #insertUsageCostStmt: Statement; @@ -5253,6 +5442,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { this.#updateStmt = this.#db.prepare( `UPDATE auth_credentials SET credential_type = ?, data = ?, identity_key = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE id = ?`, ); + this.#updateIfMatchesStmt = this.#db.prepare( + `UPDATE auth_credentials SET credential_type = ?, data = ?, identity_key = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE id = ? AND data = ? AND disabled_cause IS NULL`, + ); this.#deleteStmt = this.#db.prepare( `UPDATE auth_credentials SET disabled_cause = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE id = ?`, ); @@ -5288,6 +5480,21 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { this.#deleteExpiredCredentialBlocksStmt = this.#db.prepare( "DELETE FROM auth_credential_blocks WHERE blocked_until_ms <= ?", ); + this.#acquireCredentialRefreshLeaseStmt = this.#db.prepare( + `INSERT INTO auth_credential_refresh_leases (credential_id, owner, expires_at_ms, updated_at) + VALUES (?, ?, ?, ${SQLITE_NOW_EPOCH}) + ON CONFLICT(credential_id) DO UPDATE SET + owner = excluded.owner, + expires_at_ms = excluded.expires_at_ms, + updated_at = excluded.updated_at + WHERE auth_credential_refresh_leases.expires_at_ms <= ?`, + ); + this.#getCredentialRefreshLeaseStmt = this.#db.prepare( + "SELECT expires_at_ms FROM auth_credential_refresh_leases WHERE credential_id = ?", + ); + this.#releaseCredentialRefreshLeaseStmt = this.#db.prepare( + "DELETE FROM auth_credential_refresh_leases WHERE credential_id = ? AND owner = ?", + ); this.#insertUsageHistoryStmt = this.#db.prepare( "INSERT INTO usage_history (recorded_at, provider, account_key, email, account_id, limit_id, label, window_label, used_fraction, status, resets_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", ); @@ -5400,6 +5607,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { if (!this.#authCredentialsTableExists()) { this.#createAuthCredentialsTable(); this.#createAuthCredentialBlocksTable(); + this.#createAuthCredentialRefreshLeasesTable(); this.#writeAuthSchemaVersion(AUTH_SCHEMA_VERSION); return; } @@ -5417,6 +5625,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { this.#createAuthCredentialIndexes(); this.#createAuthCredentialBlocksTable(); + this.#createAuthCredentialRefreshLeasesTable(); this.#backfillCredentialIdentityKeys(); // Rewriting an already-current version row is a no-op write transaction // on every boot; only persist when the recorded version actually changes. @@ -5514,6 +5723,18 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { `); } + #createAuthCredentialRefreshLeasesTable(): void { + this.#db.run(` + CREATE TABLE IF NOT EXISTS auth_credential_refresh_leases ( + credential_id INTEGER PRIMARY KEY, + owner TEXT NOT NULL, + expires_at_ms INTEGER NOT NULL, + updated_at INTEGER NOT NULL + ); + CREATE INDEX IF NOT EXISTS idx_auth_credential_refresh_leases_expires ON auth_credential_refresh_leases(expires_at_ms); + `); + } + #migrateAuthSchema(fromVersion: number): void { if (fromVersion < 1) { this.#migrateAuthSchemaV0ToV1(); @@ -5527,6 +5748,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { if (fromVersion < 5) { this.#migrateAuthSchemaV4ToV5(); } + if (fromVersion < 6) { + this.#migrateAuthSchemaV5ToV6(); + } } #migrateAuthSchemaV0ToV1(): void { @@ -5620,6 +5844,13 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { migrate(); } + #migrateAuthSchemaV5ToV6(): void { + const migrate = this.#db.transaction(() => { + this.#createAuthCredentialRefreshLeasesTable(); + }); + migrate(); + } + #backfillCredentialIdentityKeys(): void { const selectRowsStmt = this.#db.prepare( "SELECT id, provider, credential_type, data, disabled_cause, identity_key FROM auth_credentials WHERE identity_key IS NULL ORDER BY id ASC", @@ -5826,6 +6057,35 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { } } + tryUpdateAuthCredentialIfMatches(id: number, expectedData: string, credential: AuthCredential): boolean { + try { + const providerStmt = this.#db.prepare("SELECT provider FROM auth_credentials WHERE id = ?"); + let providerRow: { provider?: string } | undefined; + try { + providerRow = providerStmt.get(id) as { provider?: string } | undefined; + } finally { + providerStmt.finalize(); + } + const provider = providerRow?.provider ?? ""; + const serialized = serializeCredential(provider, credential); + if (!serialized) return false; + const result = this.#updateIfMatchesStmt.run( + serialized.credentialType, + serialized.data, + serialized.identityKey, + id, + expectedData, + ) as { changes: number }; + if (result.changes !== 1) return false; + if (provider) { + this.#purgeSupersededDisabledRows(provider, this.listAuthCredentials(provider)); + } + return true; + } catch { + return false; + } + } + deleteAuthCredential(id: number, disabledCause: string): void { try { this.#deleteStmt.run(normalizeDisabledCause(disabledCause), id); @@ -5959,6 +6219,28 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { return blocks; } + tryAcquireCredentialRefreshLease(credentialId: number, owner: string, expiresAtMs: number): boolean { + const result = this.#acquireCredentialRefreshLeaseStmt.run(credentialId, owner, expiresAtMs, Date.now()) as { + changes: number; + }; + return result.changes === 1; + } + + getCredentialRefreshLeaseExpiresAt(credentialId: number): number | undefined { + const row = this.#getCredentialRefreshLeaseStmt.get(credentialId) as { expires_at_ms?: number } | undefined; + if (typeof row?.expires_at_ms !== "number") return undefined; + if (row.expires_at_ms <= Date.now()) return undefined; + return row.expires_at_ms; + } + + releaseCredentialRefreshLease(credentialId: number, owner: string): void { + try { + this.#releaseCredentialRefreshLeaseStmt.run(credentialId, owner); + } catch { + // Ignore lease release failures; expired leases are stealable. + } + } + recordUsageSnapshots(entries: UsageHistoryEntry[]): void { try { for (const entry of entries) { diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 04ae4eb4d..4dd4457cd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed concurrent MCP OAuth refreshes across OMP processes so rotating refresh tokens are refreshed once, waiters reuse the canonical credential, and stale `invalid_grant` losers cannot clear the winner. ([#5081](https://github.com/can1357/oh-my-pi/issues/5081)) + ## [16.4.0] - 2026-07-10 ### Breaking Changes diff --git a/packages/coding-agent/src/mcp/manager.ts b/packages/coding-agent/src/mcp/manager.ts index e396c9bcb..304a0d245 100644 --- a/packages/coding-agent/src/mcp/manager.ts +++ b/packages/coding-agent/src/mcp/manager.ts @@ -11,7 +11,7 @@ import { logger } from "@oh-my-pi/pi-utils"; import type { SourceMeta } from "../capability/types"; import { resolveConfigValue } from "../config/resolve-config-value"; import type { CustomTool } from "../extensibility/custom-tools/types"; -import type { AuthStorage } from "../session/auth-storage"; +import { type AuthStorage, REMOTE_REFRESH_SENTINEL } from "../session/auth-storage"; import { connectToServer, disconnectServer, @@ -1226,76 +1226,78 @@ export class MCPManager { const { credentialId } = lookup; try { let credential: MCPStoredOAuthCredential | undefined = lookup.credential; - // Refresh material comes from ONE source: the credential's embedded - // fields (written atomically with the tokens they minted — tokenUrl - // always present) or, for legacy rows that predate embedding, the - // config auth block. Never mix the two: a shared file's auth block - // can belong to another profile, whose client the grant is NOT - // bound to. - const material = selectMcpOAuthRefreshMaterial(credential, auth); - const tokenUrl = material?.tokenUrl; - const clientId = material?.clientId; - const clientSecret = material?.clientSecret; - // `authorizationUrl` only lives on the embedded credential form; - // legacy `MCPAuthConfig` rows never carried it. Required to filter - // same-origin resource indicators on refresh when the authorize and - // token endpoints sit on different origins (issue #3502 review - // follow-up). - const authorizationUrl = material && "authorizationUrl" in material ? material.authorizationUrl : undefined; - const resourceIsFallback = - !material?.resource && (config.type === "http" || config.type === "sse") && Boolean(config.url); - const resource = material?.resource ?? (resourceIsFallback ? config.url : undefined); - // Proactive refresh: 5-minute buffer before expiry - // Force refresh: on 401/403 auth errors (revoked tokens, clock skew, missing expires) const REFRESH_BUFFER_MS = 5 * 60_000; - const shouldRefresh = - opts?.forceRefresh || (credential.expires && Date.now() >= credential.expires - REFRESH_BUFFER_MS); - if (shouldRefresh && credential.refresh && tokenUrl) { - try { - const refreshed = await refreshMCPOAuthToken( - tokenUrl, - credential.refresh, - clientId, - clientSecret, - resource, - { authorizationUrl, stripSameOriginResource: resourceIsFallback }, - ); - // Spread the old credential first so embedded refresh material survives rotation. - const refreshedCredential: MCPStoredOAuthCredential = { - ...credential, - ...refreshed, - tokenUrl, - clientId, - clientSecret, - resource: resourceIsFallback ? undefined : resource, - authorizationUrl, - }; - await this.#authStorage.set(credentialId, refreshedCredential); - credential = refreshedCredential; - } catch (refreshError) { - const errorMsg = refreshError instanceof Error ? refreshError.message : String(refreshError); - if (isDefinitiveOAuthFailure(errorMsg)) { - // `invalid_grant` / `invalid_token` / 401 from the token endpoint means - // the server has retired this credential — keeping the stale access - // token would just re-fail with 401 on every MCP request and leave a - // poisoned row in agent.db that survives restarts. Drop it now so the - // next connect attempt surfaces a clean "needs reauth" failure and - // the user can recover with `/mcp reauth ` (or `/mcp unauth` - // to forget the server entirely). - logger.warn("MCP OAuth refresh failed definitively; cleared credential", { - credentialId, - error: errorMsg, + const refreshResult = await this.#authStorage.refreshStoredOAuthCredential( + credentialId, + { + observedCredential: credential, + credentialFromRow: row => row, + forceRefresh: opts?.forceRefresh, + refreshSkewMs: REFRESH_BUFFER_MS, + canRefresh: current => { + const material = selectMcpOAuthRefreshMaterial(current, auth); + return Boolean(current.refresh && material?.tokenUrl); + }, + refresh: current => { + if (current.refresh === REMOTE_REFRESH_SENTINEL) { + throw new Error("MCP OAuth refresh token is broker-redacted; local refresh is unavailable"); + } + const material = selectMcpOAuthRefreshMaterial(current, auth); + const tokenUrl = material?.tokenUrl; + if (!current.refresh || !tokenUrl) { + throw new Error("MCP OAuth credential is missing refresh material"); + } + const clientId = material?.clientId; + const clientSecret = material?.clientSecret; + const authorizationUrl = + material && "authorizationUrl" in material ? material.authorizationUrl : undefined; + const resourceIsFallback = + !material?.resource && (config.type === "http" || config.type === "sse") && Boolean(config.url); + const resource = material?.resource ?? (resourceIsFallback ? config.url : undefined); + return refreshMCPOAuthToken(tokenUrl, current.refresh, clientId, clientSecret, resource, { + authorizationUrl, + stripSameOriginResource: resourceIsFallback, }); - await this.#authStorage.remove(credentialId); - credential = undefined; - } else { + }, + mergeRefreshedCredential: (current, refreshed) => { + const material = selectMcpOAuthRefreshMaterial(current, auth); + const tokenUrl = material?.tokenUrl; + const clientId = material?.clientId; + const clientSecret = material?.clientSecret; + const authorizationUrl = + material && "authorizationUrl" in material ? material.authorizationUrl : undefined; + const resourceIsFallback = + !material?.resource && (config.type === "http" || config.type === "sse") && Boolean(config.url); + const resource = material?.resource ?? (resourceIsFallback ? config.url : undefined); + return { + ...current, + ...refreshed, + tokenUrl, + clientId, + clientSecret, + resource: resourceIsFallback ? undefined : resource, + authorizationUrl, + }; + }, + isDefinitiveFailure: error => + isDefinitiveOAuthFailure(error instanceof Error ? error.message : String(error)), + disabledCause: error => + `oauth refresh failed: ${error instanceof Error ? error.message : String(error)}`, + keepCredentialOnRefreshFailure: error => + !(error instanceof Error && error.message.includes("broker-redacted")), + onRefreshFailure: refreshError => { + if (refreshError instanceof Error && refreshError.message.includes("broker-redacted")) return; logger.warn("MCP OAuth refresh failed, using existing token", { credentialId, error: refreshError, }); - } - } + }, + }, + ); + if (refreshResult.removed) { + logger.warn("MCP OAuth refresh failed definitively; cleared credential", { credentialId }); } + credential = refreshResult.credential; if (credential) { if (resolved.type === "http" || resolved.type === "sse") { diff --git a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts index c2f7c1ecc..7c5f1521e 100644 --- a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts +++ b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts @@ -11,22 +11,59 @@ */ import { Database } from "bun:sqlite"; import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai"; import { MCPManager } from "@oh-my-pi/pi-coding-agent/mcp/manager"; import * as oauthFlow from "@oh-my-pi/pi-coding-agent/mcp/oauth-flow"; import type { MCPServerConfig } from "@oh-my-pi/pi-coding-agent/mcp/types"; +import { removeWithRetries } from "@oh-my-pi/pi-utils"; const CREDENTIAL_ID = "mcp_oauth_test_1908"; const TOKEN_URL = "https://example.com/oauth/token"; const STALE_ACCESS = "stale-access-token"; const STALE_REFRESH = "stale-refresh-token"; +const SHARED_CREDENTIAL_ID = "mcp_oauth_test_5081"; +const SHARED_STALE_ACCESS = "access-0"; +const SHARED_STALE_REFRESH = "refresh-0"; +const SHARED_FRESH_ACCESS = "access-1"; +const SHARED_FRESH_REFRESH = "refresh-1"; + /** Build a `Headers` snapshot from a prepared MCP config. */ function getAuthorizationHeader(config: MCPServerConfig): string | undefined { if (config.type !== "http" && config.type !== "sse") return undefined; return config.headers?.Authorization; } +async function withSharedSQLiteAuth( + fn: (context: { + authA: AuthStorage; + authB: AuthStorage; + storeA: SqliteAuthCredentialStore; + storeB: SqliteAuthCredentialStore; + }) => Promise, +): Promise { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-mcp-oauth-shared-refresh-")); + let authA: AuthStorage | undefined; + let authB: AuthStorage | undefined; + try { + const dbPath = path.join(tempDir, "agent.db"); + const storeA = await SqliteAuthCredentialStore.open(dbPath); + const storeB = await SqliteAuthCredentialStore.open(dbPath); + authA = new AuthStorage(storeA); + authB = new AuthStorage(storeB); + await authA.reload(); + await authB.reload(); + return await fn({ authA, authB, storeA, storeB }); + } finally { + authA?.close(); + authB?.close(); + await removeWithRetries(tempDir); + } +} + describe("MCPManager OAuth refresh failure", () => { let manager: MCPManager; let authStorage: AuthStorage; @@ -137,3 +174,156 @@ describe("MCPManager OAuth refresh failure", () => { expect(remaining).toMatchObject({ type: "oauth", access: "fresh-access", refresh: "fresh-refresh" }); }); }); + +describe("MCPManager shared SQLite OAuth refresh", () => { + test("shares refresh ownership so peer managers do not replay a rotating refresh token", async () => { + await withSharedSQLiteAuth(async ({ authA, authB }) => { + await authA.set(SHARED_CREDENTIAL_ID, { + type: "oauth", + access: SHARED_STALE_ACCESS, + refresh: SHARED_STALE_REFRESH, + expires: Date.now() - 60_000, + }); + await authB.reload(); + + const refreshStarted = Promise.withResolvers(); + const allowRefreshResponse = Promise.withResolvers(); + const refreshTokens: string[] = []; + let refreshRequests = 0; + const tokenServer = Bun.serve({ + hostname: "127.0.0.1", + port: 0, + async fetch(req) { + if (req.method !== "POST" || new URL(req.url).pathname !== "/token") { + return new Response("not found", { status: 404 }); + } + refreshRequests += 1; + const body = new URLSearchParams(await req.text()); + refreshTokens.push(body.get("refresh_token") ?? ""); + if (refreshRequests === 1) { + refreshStarted.resolve(); + await allowRefreshResponse.promise; + return Response.json({ + access_token: SHARED_FRESH_ACCESS, + refresh_token: SHARED_FRESH_REFRESH, + expires_in: 3600, + }); + } + return Response.json({ error: "invalid_grant" }, { status: 400 }); + }, + }); + try { + const managerA = new MCPManager(process.cwd()); + managerA.setAuthStorage(authA); + const managerB = new MCPManager(process.cwd()); + managerB.setAuthStorage(authB); + const config: MCPServerConfig = { + type: "http", + url: "https://logfire.example.com/mcp", + auth: { + type: "oauth", + credentialId: SHARED_CREDENTIAL_ID, + tokenUrl: `http://127.0.0.1:${tokenServer.port}/token`, + }, + }; + + const preparedA = managerA.prepareConfig(config); + await refreshStarted.promise; + const preparedB = managerB.prepareConfig(config); + allowRefreshResponse.resolve(); + + const [resolvedA, resolvedB] = await Promise.all([preparedA, preparedB]); + + expect(refreshRequests).toBe(1); + expect(refreshTokens).toEqual([SHARED_STALE_REFRESH]); + expect(getAuthorizationHeader(resolvedA)).toBe(`Bearer ${SHARED_FRESH_ACCESS}`); + expect(getAuthorizationHeader(resolvedB)).toBe(`Bearer ${SHARED_FRESH_ACCESS}`); + + await authA.reload(); + const canonical = authA.get(SHARED_CREDENTIAL_ID); + expect(canonical).toMatchObject({ + type: "oauth", + access: SHARED_FRESH_ACCESS, + refresh: SHARED_FRESH_REFRESH, + }); + } finally { + tokenServer.stop(true); + } + }); + }); + + test("keeps the peer-rotated credential when a stale refresh attempt returns invalid_grant", async () => { + await withSharedSQLiteAuth(async ({ authA, authB, storeB }) => { + await authA.set(SHARED_CREDENTIAL_ID, { + type: "oauth", + access: SHARED_STALE_ACCESS, + refresh: SHARED_STALE_REFRESH, + expires: Date.now() - 60_000, + }); + await authB.reload(); + const storedBefore = storeB.listAuthCredentials(SHARED_CREDENTIAL_ID); + expect(storedBefore).toHaveLength(1); + const rowId = storedBefore[0]!.id; + + const refreshStarted = Promise.withResolvers(); + const allowInvalidGrant = Promise.withResolvers(); + const refreshTokens: string[] = []; + let refreshRequests = 0; + const tokenServer = Bun.serve({ + hostname: "127.0.0.1", + port: 0, + async fetch(req) { + if (req.method !== "POST" || new URL(req.url).pathname !== "/token") { + return new Response("not found", { status: 404 }); + } + refreshRequests += 1; + const body = new URLSearchParams(await req.text()); + refreshTokens.push(body.get("refresh_token") ?? ""); + refreshStarted.resolve(); + await allowInvalidGrant.promise; + return Response.json({ error: "invalid_grant" }, { status: 400 }); + }, + }); + try { + const managerA = new MCPManager(process.cwd()); + managerA.setAuthStorage(authA); + const config: MCPServerConfig = { + type: "http", + url: "https://logfire.example.com/mcp", + auth: { + type: "oauth", + credentialId: SHARED_CREDENTIAL_ID, + tokenUrl: `http://127.0.0.1:${tokenServer.port}/token`, + }, + }; + + const prepared = managerA.prepareConfig(config); + await refreshStarted.promise; + storeB.updateAuthCredential(rowId, { + type: "oauth", + access: SHARED_FRESH_ACCESS, + refresh: SHARED_FRESH_REFRESH, + expires: Date.now() + 60 * 60_000, + }); + allowInvalidGrant.resolve(); + + const resolved = await prepared; + + expect(refreshRequests).toBe(1); + expect(refreshTokens).toEqual([SHARED_STALE_REFRESH]); + expect(getAuthorizationHeader(resolved)).toBe(`Bearer ${SHARED_FRESH_ACCESS}`); + + await authA.reload(); + const canonical = authA.get(SHARED_CREDENTIAL_ID); + expect(canonical).toMatchObject({ + type: "oauth", + access: SHARED_FRESH_ACCESS, + refresh: SHARED_FRESH_REFRESH, + }); + expect(storeB.listAuthCredentials(SHARED_CREDENTIAL_ID)).toHaveLength(1); + } finally { + tokenServer.stop(true); + } + }); + }); +}); From cb3285313006da9366e1d9454281a78826795a30 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 21:15:04 +0000 Subject: [PATCH 096/205] fix(auth): renewed oauth refresh leases - Renewed durable OAuth refresh leases while token refreshes are in flight so slow endpoints cannot let a peer steal the row and replay a rotating refresh token. - Let CAS update and disable storage errors propagate instead of collapsing them into peer-win misses. - Added regressions for lease renewal and CAS storage failure propagation. Fixes #5081 --- packages/ai/src/auth-storage.ts | 167 +++++++++++------- .../auth-storage-oauth-refresh-race.test.ts | 78 ++++++++ .../test/mcp-manager-oauth-refresh.test.ts | 142 ++++++++++++++- 3 files changed, 322 insertions(+), 65 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 20b777eea..ae04587ca 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -336,6 +336,7 @@ export interface AuthCredentialStore { tryAcquireCredentialRefreshLease?(credentialId: number, owner: string, expiresAtMs: number): boolean; getCredentialRefreshLeaseExpiresAt?(credentialId: number): number | undefined; releaseCredentialRefreshLease?(credentialId: number, owner: string): void; + renewCredentialRefreshLease?(credentialId: number, owner: string, expiresAtMs: number): boolean; /** * Append usage-limit snapshots for trend history. Optional: stores without * durable storage (e.g. the broker remote store) omit it and recording is @@ -599,6 +600,7 @@ const DEFAULT_OAUTH_REFRESH_TIMEOUT_MS = 10_000; const OAUTH_REFRESH_SKEW_MS = 60_000; const OAUTH_REFRESH_LEASE_TTL_MS = 15_000; const OAUTH_REFRESH_LEASE_POLL_MS = 50; +const OAUTH_REFRESH_LEASE_RENEW_MS = 5_000; /** * Cap on the buffered credential_disabled backlog held while no handler is attached. * In practice the backlog is 0–N where N ≈ active providers (≤ ~20). The cap exists so @@ -1873,7 +1875,8 @@ export class AuthStorage { const hasDurableLease = !!this.#store.tryAcquireCredentialRefreshLease && !!this.#store.getCredentialRefreshLeaseExpiresAt && - !!this.#store.releaseCredentialRefreshLease; + !!this.#store.releaseCredentialRefreshLease && + !!this.#store.renewCredentialRefreshLease; const owner = crypto.randomUUID(); let leasedCredentialId: number | undefined; @@ -1943,42 +1946,76 @@ export class AuthStorage { const serialized = serializeCredential(provider, current); if (!serialized) return { credential: current, refreshed: false, removed: false }; + let stopLeaseRenewal = false; + let leaseRenewalError: unknown; + const leaseRenewalStopped = Promise.withResolvers(); + const leaseRenewal = + leasedCredentialId !== undefined + ? (async () => { + while (!stopLeaseRenewal) { + await Promise.race([Bun.sleep(OAUTH_REFRESH_LEASE_RENEW_MS), leaseRenewalStopped.promise]); + if (stopLeaseRenewal) return; + const renewed = this.#store.renewCredentialRefreshLease?.( + leasedCredentialId, + owner, + Date.now() + OAUTH_REFRESH_LEASE_TTL_MS, + ); + if (!renewed) { + throw new AIError.ConfigurationError("OAuth refresh ownership was lost before persistence"); + } + } + })().catch(error => { + leaseRenewalError = error; + }) + : undefined; + let refreshed: OAuthCredentials; try { - refreshed = await options.refresh(current); - } catch (error) { - if (options.isDefinitiveFailure?.(error)) { - const disabledCause = options.disabledCause?.(error) ?? `oauth refresh failed: ${String(error)}`; - const disabled = this.#store.tryDisableAuthCredentialIfMatches(row.id, serialized.data, disabledCause); - if (disabled) { - this.#setStoredCredentials( - provider, - rows - .filter(entry => entry.id !== row.id) - .map(entry => ({ id: entry.id, credential: entry.credential })), + try { + refreshed = await options.refresh(current); + } catch (error) { + if (options.isDefinitiveFailure?.(error)) { + const disabledCause = options.disabledCause?.(error) ?? `oauth refresh failed: ${String(error)}`; + const disabled = this.#store.tryDisableAuthCredentialIfMatches( + row.id, + serialized.data, + disabledCause, ); - this.#resetProviderAssignments(provider); - this.#emitCredentialDisabled({ provider, disabledCause }); - return { credential: undefined, refreshed: false, removed: true }; + if (disabled) { + this.#setStoredCredentials( + provider, + rows + .filter(entry => entry.id !== row.id) + .map(entry => ({ id: entry.id, credential: entry.credential })), + ); + this.#resetProviderAssignments(provider); + this.#emitCredentialDisabled({ provider, disabledCause }); + return { credential: undefined, refreshed: false, removed: true }; + } + await this.reload(); + const latest = this.get(provider); + return { + credential: latest?.type === "oauth" ? options.credentialFromRow(latest) : undefined, + refreshed: false, + removed: false, + }; } - await this.reload(); - const latest = this.get(provider); - return { - credential: latest?.type === "oauth" ? options.credentialFromRow(latest) : undefined, - refreshed: false, - removed: false, - }; + options.onRefreshFailure?.(error); + const keepCredential = + typeof options.keepCredentialOnRefreshFailure === "function" + ? options.keepCredentialOnRefreshFailure(error) + : options.keepCredentialOnRefreshFailure === true; + if (keepCredential) { + return { credential: current, refreshed: false, removed: false }; + } + throw error; } - options.onRefreshFailure?.(error); - const keepCredential = - typeof options.keepCredentialOnRefreshFailure === "function" - ? options.keepCredentialOnRefreshFailure(error) - : options.keepCredentialOnRefreshFailure === true; - if (keepCredential) { - return { credential: current, refreshed: false, removed: false }; - } - throw error; + } finally { + stopLeaseRenewal = true; + leaseRenewalStopped.resolve(); + await leaseRenewal; } + if (leaseRenewalError) throw leaseRenewalError; const merged: T = options.mergeRefreshedCredential ? options.mergeRefreshedCredential(current, refreshed) @@ -5413,6 +5450,7 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { #deleteExpiredCredentialBlocksStmt: Statement; #acquireCredentialRefreshLeaseStmt: Statement; #getCredentialRefreshLeaseStmt: Statement; + #renewCredentialRefreshLeaseStmt: Statement; #releaseCredentialRefreshLeaseStmt: Statement; #credentialBlockReconcileAfter: Map = new Map(); #insertUsageHistoryStmt: Statement; @@ -5492,6 +5530,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { this.#getCredentialRefreshLeaseStmt = this.#db.prepare( "SELECT expires_at_ms FROM auth_credential_refresh_leases WHERE credential_id = ?", ); + this.#renewCredentialRefreshLeaseStmt = this.#db.prepare( + `UPDATE auth_credential_refresh_leases SET expires_at_ms = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE credential_id = ? AND owner = ?`, + ); this.#releaseCredentialRefreshLeaseStmt = this.#db.prepare( "DELETE FROM auth_credential_refresh_leases WHERE credential_id = ? AND owner = ?", ); @@ -6058,32 +6099,28 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { } tryUpdateAuthCredentialIfMatches(id: number, expectedData: string, credential: AuthCredential): boolean { + const providerStmt = this.#db.prepare("SELECT provider FROM auth_credentials WHERE id = ?"); + let providerRow: { provider?: string } | undefined; try { - const providerStmt = this.#db.prepare("SELECT provider FROM auth_credentials WHERE id = ?"); - let providerRow: { provider?: string } | undefined; - try { - providerRow = providerStmt.get(id) as { provider?: string } | undefined; - } finally { - providerStmt.finalize(); - } - const provider = providerRow?.provider ?? ""; - const serialized = serializeCredential(provider, credential); - if (!serialized) return false; - const result = this.#updateIfMatchesStmt.run( - serialized.credentialType, - serialized.data, - serialized.identityKey, - id, - expectedData, - ) as { changes: number }; - if (result.changes !== 1) return false; - if (provider) { - this.#purgeSupersededDisabledRows(provider, this.listAuthCredentials(provider)); - } - return true; - } catch { - return false; + providerRow = providerStmt.get(id) as { provider?: string } | undefined; + } finally { + providerStmt.finalize(); } + const provider = providerRow?.provider ?? ""; + const serialized = serializeCredential(provider, credential); + if (!serialized) return false; + const result = this.#updateIfMatchesStmt.run( + serialized.credentialType, + serialized.data, + serialized.identityKey, + id, + expectedData, + ) as { changes: number }; + if (result.changes !== 1) return false; + if (provider) { + this.#purgeSupersededDisabledRows(provider, this.listAuthCredentials(provider)); + } + return true; } deleteAuthCredential(id: number, disabledCause: string): void { @@ -6101,16 +6138,11 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { * row between our pre-check and the disable. */ tryDisableAuthCredentialIfMatches(id: number, expectedData: string, disabledCause: string): boolean { - try { - const result = this.#deleteIfMatchesStmt.run(normalizeDisabledCause(disabledCause), id, expectedData) as { - changes: number; - }; - return result.changes === 1; - } catch { - return false; - } + const result = this.#deleteIfMatchesStmt.run(normalizeDisabledCause(disabledCause), id, expectedData) as { + changes: number; + }; + return result.changes === 1; } - deleteAuthCredentialsForProvider(provider: string, disabledCause: string): void { try { this.#deleteByProviderStmt.run(normalizeDisabledCause(disabledCause), provider); @@ -6233,6 +6265,13 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { return row.expires_at_ms; } + renewCredentialRefreshLease(credentialId: number, owner: string, expiresAtMs: number): boolean { + const result = this.#renewCredentialRefreshLeaseStmt.run(expiresAtMs, credentialId, owner) as { + changes: number; + }; + return result.changes === 1; + } + releaseCredentialRefreshLease(credentialId: number, owner: string): void { try { this.#releaseCredentialRefreshLeaseStmt.run(credentialId, owner); diff --git a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts index 6b9323958..b5179e450 100644 --- a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts +++ b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts @@ -432,4 +432,82 @@ describe("AuthStorage OAuth refresh race", () => { expect(cRow?.credential.type).toBe("oauth"); if (cRow?.credential.type === "oauth") expect(cRow.credential.refresh).toBe("c-ref"); }); + + test("propagates CAS update storage errors instead of treating them as peer refresh wins", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + + await authStorage.set("unit-oauth-cas-update-error", [ + { + type: "oauth", + access: "access-old", + refresh: "refresh-old", + expires: Date.now() - 60_000, + }, + ]); + + const failure = new Error("sqlite update failed"); + vi.spyOn(store, "tryUpdateAuthCredentialIfMatches").mockImplementation(() => { + throw failure; + }); + + await expect( + authStorage.refreshStoredOAuthCredential("unit-oauth-cas-update-error", { + credentialFromRow: row => row, + forceRefresh: true, + refresh: async credential => ({ + ...credential, + access: "access-fresh", + refresh: "refresh-fresh", + expires: Date.now() + 60 * 60_000, + }), + }), + ).rejects.toThrow("sqlite update failed"); + + const stored = store.listAuthCredentials("unit-oauth-cas-update-error"); + expect(stored).toHaveLength(1); + expect(stored[0]?.credential).toMatchObject({ + type: "oauth", + access: "access-old", + refresh: "refresh-old", + }); + }); + + test("propagates CAS disable storage errors instead of treating them as peer rotations", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + + await authStorage.set("unit-oauth-cas-disable-error", [ + { + type: "oauth", + access: "access-old", + refresh: "refresh-old", + expires: Date.now() - 60_000, + }, + ]); + + const failure = new Error("sqlite disable failed"); + vi.spyOn(store, "tryDisableAuthCredentialIfMatches").mockImplementation(() => { + throw failure; + }); + + await expect( + authStorage.refreshStoredOAuthCredential("unit-oauth-cas-disable-error", { + credentialFromRow: row => row, + forceRefresh: true, + refresh: async () => { + throw new Error('HTTP 400 invalid_grant {"error":"invalid_grant"}'); + }, + isDefinitiveFailure: error => error instanceof Error && error.message.includes("invalid_grant"), + disabledCause: error => `oauth refresh failed: ${error instanceof Error ? error.message : String(error)}`, + }), + ).rejects.toThrow("sqlite disable failed"); + + expect(events).toHaveLength(0); + const stored = store.listAuthCredentials("unit-oauth-cas-disable-error"); + expect(stored).toHaveLength(1); + expect(stored[0]?.credential).toMatchObject({ + type: "oauth", + access: "access-old", + refresh: "refresh-old", + }); + }); }); diff --git a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts index 7c5f1521e..8ad061a0f 100644 --- a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts +++ b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts @@ -10,7 +10,7 @@ * Bearer injection, so the next request surfaces a clean auth error instead. */ import { Database } from "bun:sqlite"; -import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, setSystemTime, test, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; @@ -37,6 +37,54 @@ function getAuthorizationHeader(config: MCPServerConfig): string | undefined { return config.headers?.Authorization; } +type ControlledSleep = { + ms: number; + resolved: boolean; + resolve: () => void; +}; + +function installControlledBunSleep(): ControlledSleep[] { + const calls: ControlledSleep[] = []; + vi.spyOn(Bun, "sleep").mockImplementation((ms: number | Date) => { + const { promise, resolve } = Promise.withResolvers(); + let call: ControlledSleep; + const delayMs = typeof ms === "number" ? ms : Math.max(0, ms.getTime() - Date.now()); + call = { + ms: delayMs, + resolved: false, + resolve: () => { + if (call.resolved) return; + call.resolved = true; + resolve(); + }, + }; + calls.push(call); + return promise; + }); + return calls; +} + +async function drainMicrotasks(count = 10): Promise { + for (let attempt = 0; attempt < count; attempt++) { + await Promise.resolve(); + } +} + +async function waitForControlledSleep(calls: ControlledSleep[], ms: number): Promise { + for (let attempt = 0; attempt < 20; attempt++) { + const call = calls.find(candidate => !candidate.resolved && candidate.ms === ms); + if (call) return call; + await Promise.resolve(); + } + throw new Error(`Timed out waiting for Bun.sleep(${ms})`); +} + +function resolvePendingControlledSleeps(calls: ControlledSleep[]): void { + for (const call of calls) { + if (!call.resolved) call.resolve(); + } +} + async function withSharedSQLiteAuth( fn: (context: { authA: AuthStorage; @@ -176,6 +224,98 @@ describe("MCPManager OAuth refresh failure", () => { }); describe("MCPManager shared SQLite OAuth refresh", () => { + afterEach(() => { + setSystemTime(); + vi.restoreAllMocks(); + }); + + test("renews refresh ownership while the token endpoint is blocked", async () => { + await withSharedSQLiteAuth(async ({ authA, authB, storeA }) => { + const startMs = Date.parse("2026-07-10T12:00:00.000Z"); + setSystemTime(new Date(startMs)); + const sleeps = installControlledBunSleep(); + const renewSpy = vi.spyOn(storeA, "renewCredentialRefreshLease"); + + await authA.set(SHARED_CREDENTIAL_ID, { + type: "oauth", + access: SHARED_STALE_ACCESS, + refresh: SHARED_STALE_REFRESH, + expires: startMs - 60_000, + }); + await authB.reload(); + + const refreshStarted = Promise.withResolvers(); + const allowRefreshResponse = Promise.withResolvers(); + const refreshTokens: string[] = []; + let refreshRequests = 0; + const tokenServer = Bun.serve({ + hostname: "127.0.0.1", + port: 0, + async fetch(req) { + if (req.method !== "POST" || new URL(req.url).pathname !== "/token") { + return new Response("not found", { status: 404 }); + } + refreshRequests += 1; + const body = new URLSearchParams(await req.text()); + refreshTokens.push(body.get("refresh_token") ?? ""); + if (refreshRequests === 1) { + refreshStarted.resolve(); + await allowRefreshResponse.promise; + return Response.json({ + access_token: SHARED_FRESH_ACCESS, + refresh_token: SHARED_FRESH_REFRESH, + expires_in: 3600, + }); + } + return Response.json({ error: "invalid_grant" }, { status: 400 }); + }, + }); + try { + const managerA = new MCPManager(process.cwd()); + managerA.setAuthStorage(authA); + const managerB = new MCPManager(process.cwd()); + managerB.setAuthStorage(authB); + const config: MCPServerConfig = { + type: "http", + url: "https://logfire.example.com/mcp", + auth: { + type: "oauth", + credentialId: SHARED_CREDENTIAL_ID, + tokenUrl: `http://127.0.0.1:${tokenServer.port}/token`, + }, + }; + + const preparedA = managerA.prepareConfig(config); + await refreshStarted.promise; + const renewalSleep = await waitForControlledSleep(sleeps, 5_000); + + setSystemTime(new Date(startMs + 5_000)); + renewalSleep.resolve(); + await drainMicrotasks(); + expect(renewSpy).toHaveBeenCalledTimes(1); + + setSystemTime(new Date(startMs + 16_000)); + const preparedB = managerB.prepareConfig(config); + const peerLeaseWait = await waitForControlledSleep(sleeps, 250); + + expect(refreshRequests).toBe(1); + expect(refreshTokens).toEqual([SHARED_STALE_REFRESH]); + + allowRefreshResponse.resolve(); + const resolvedA = await preparedA; + peerLeaseWait.resolve(); + const resolvedB = await preparedB; + + expect(getAuthorizationHeader(resolvedA)).toBe(`Bearer ${SHARED_FRESH_ACCESS}`); + expect(getAuthorizationHeader(resolvedB)).toBe(`Bearer ${SHARED_FRESH_ACCESS}`); + expect(refreshRequests).toBe(1); + expect(refreshTokens).toEqual([SHARED_STALE_REFRESH]); + } finally { + resolvePendingControlledSleeps(sleeps); + tokenServer.stop(true); + } + }); + }); test("shares refresh ownership so peer managers do not replay a rotating refresh token", async () => { await withSharedSQLiteAuth(async ({ authA, authB }) => { await authA.set(SHARED_CREDENTIAL_ID, { From b60dc669ea2974efbfd356b1e72a8d25ff540bf5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 21:49:18 +0000 Subject: [PATCH 097/205] fix(auth): fenced oauth refresh writes - Fenced final OAuth refresh update and terminal-disable CAS statements by row id, serialized credential data, active lease owner, and unexpired lease time. - Passed an AbortSignal through MCP OAuth token refresh and bounded owned refresh operations below the lease TTL while awaiting the aborted fetch to settle. - Added regressions for stolen-lease update/disable attempts and timed-out MCP token fetch abort behavior. Fixes #5081 --- packages/ai/src/auth-storage.ts | 119 +++++++++++++++--- .../auth-storage-oauth-refresh-race.test.ts | 109 +++++++++++++++- packages/coding-agent/src/mcp/manager.ts | 3 +- packages/coding-agent/src/mcp/oauth-flow.ts | 2 + .../test/mcp-manager-oauth-refresh.test.ts | 60 ++++++++- 5 files changed, 272 insertions(+), 21 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index ae04587ca..60e535093 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -308,13 +308,28 @@ export interface AuthCredentialSnapshot { * a remote broker; mutating methods (`replace*`, `upsert*`, `delete*ForProvider`) * throw because login flows route through the broker, not the client. */ +export interface CredentialRefreshLeaseFence { + owner: string; + nowMs: number; +} + export interface AuthCredentialStore { close(): void; listAuthCredentials(provider?: string): StoredAuthCredential[]; updateAuthCredential(id: number, credential: AuthCredential): void; deleteAuthCredential(id: number, disabledCause: string): void; - tryDisableAuthCredentialIfMatches(id: number, expectedData: string, disabledCause: string): boolean; - tryUpdateAuthCredentialIfMatches?(id: number, expectedData: string, credential: AuthCredential): boolean; + tryDisableAuthCredentialIfMatches( + id: number, + expectedData: string, + disabledCause: string, + lease?: CredentialRefreshLeaseFence, + ): boolean; + tryUpdateAuthCredentialIfMatches?( + id: number, + expectedData: string, + credential: AuthCredential, + lease?: CredentialRefreshLeaseFence, + ): boolean; replaceAuthCredentialsForProvider(provider: string, credentials: AuthCredential[]): StoredAuthCredential[]; upsertAuthCredentialForProvider(provider: string, credential: AuthCredential): StoredAuthCredential[]; deleteAuthCredentialsForProvider(provider: string, disabledCause: string): void; @@ -601,6 +616,7 @@ const OAUTH_REFRESH_SKEW_MS = 60_000; const OAUTH_REFRESH_LEASE_TTL_MS = 15_000; const OAUTH_REFRESH_LEASE_POLL_MS = 50; const OAUTH_REFRESH_LEASE_RENEW_MS = 5_000; +const OAUTH_REFRESH_OPERATION_TIMEOUT_MS = 10_000; /** * Cap on the buffered credential_disabled backlog held while no handler is attached. * In practice the backlog is 0–N where N ≈ active providers (≤ ~20). The cap exists so @@ -736,7 +752,8 @@ export interface StoredOAuthRefreshOptions boolean); onRefreshFailure?: (error: unknown) => void; - refresh: (credential: T) => Promise; + refreshTimeoutMs?: number; + refresh: (credential: T, signal?: AbortSignal) => Promise; mergeRefreshedCredential?: (credential: T, refreshed: OAuthCredentials) => T; isDefinitiveFailure?: (error: unknown) => boolean; disabledCause?: (error: unknown) => string; @@ -1968,11 +1985,20 @@ export class AuthStorage { leaseRenewalError = error; }) : undefined; + const refreshAbort = new AbortController(); + const refreshTimeout = setTimeout(() => { + refreshAbort.abort( + new AIError.OAuthError(`OAuth token refresh timed out for provider: ${provider}`, { + kind: "timeout", + provider, + }), + ); + }, options.refreshTimeoutMs ?? OAUTH_REFRESH_OPERATION_TIMEOUT_MS); let refreshed: OAuthCredentials; try { try { - refreshed = await options.refresh(current); + refreshed = await options.refresh(current, refreshAbort.signal); } catch (error) { if (options.isDefinitiveFailure?.(error)) { const disabledCause = options.disabledCause?.(error) ?? `oauth refresh failed: ${String(error)}`; @@ -1980,6 +2006,7 @@ export class AuthStorage { row.id, serialized.data, disabledCause, + leasedCredentialId !== undefined ? { owner, nowMs: Date.now() } : undefined, ); if (disabled) { this.#setStoredCredentials( @@ -2014,6 +2041,7 @@ export class AuthStorage { stopLeaseRenewal = true; leaseRenewalStopped.resolve(); await leaseRenewal; + clearTimeout(refreshTimeout); } if (leaseRenewalError) throw leaseRenewalError; @@ -2031,7 +2059,14 @@ export class AuthStorage { apiEndpoint: refreshed.apiEndpoint ?? current.apiEndpoint, }; if (this.#store.tryUpdateAuthCredentialIfMatches) { - if (!this.#store.tryUpdateAuthCredentialIfMatches(row.id, serialized.data, merged)) { + if ( + !this.#store.tryUpdateAuthCredentialIfMatches( + row.id, + serialized.data, + merged, + leasedCredentialId !== undefined ? { owner, nowMs: Date.now() } : undefined, + ) + ) { await this.reload(); const latest = this.get(provider); return { @@ -5443,6 +5478,8 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { #getCacheIncludingExpiredStmt: Statement; #upsertCacheStmt: Statement; #deleteExpiredCacheStmt: Statement; + #updateIfMatchesWithLeaseStmt: Statement; + #deleteIfMatchesWithLeaseStmt: Statement; #getCredentialBlockStmt: Statement; #listCredentialBlocksByCredentialStmt: Statement; #upsertCredentialBlockStmt: Statement; @@ -5483,12 +5520,30 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { this.#updateIfMatchesStmt = this.#db.prepare( `UPDATE auth_credentials SET credential_type = ?, data = ?, identity_key = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE id = ? AND data = ? AND disabled_cause IS NULL`, ); + this.#updateIfMatchesWithLeaseStmt = this.#db.prepare( + `UPDATE auth_credentials + SET credential_type = ?, data = ?, identity_key = ?, updated_at = ${SQLITE_NOW_EPOCH} + WHERE id = ? AND data = ? AND disabled_cause IS NULL + AND EXISTS ( + SELECT 1 FROM auth_credential_refresh_leases + WHERE credential_id = ? AND owner = ? AND expires_at_ms > ? + )`, + ); this.#deleteStmt = this.#db.prepare( `UPDATE auth_credentials SET disabled_cause = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE id = ?`, ); this.#deleteIfMatchesStmt = this.#db.prepare( `UPDATE auth_credentials SET disabled_cause = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE id = ? AND data = ? AND disabled_cause IS NULL`, ); + this.#deleteIfMatchesWithLeaseStmt = this.#db.prepare( + `UPDATE auth_credentials + SET disabled_cause = ?, updated_at = ${SQLITE_NOW_EPOCH} + WHERE id = ? AND data = ? AND disabled_cause IS NULL + AND EXISTS ( + SELECT 1 FROM auth_credential_refresh_leases + WHERE credential_id = ? AND owner = ? AND expires_at_ms > ? + )`, + ); this.#deleteByProviderStmt = this.#db.prepare( `UPDATE auth_credentials SET disabled_cause = ?, updated_at = ${SQLITE_NOW_EPOCH} WHERE provider = ? AND disabled_cause IS NULL`, ); @@ -6098,7 +6153,12 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { } } - tryUpdateAuthCredentialIfMatches(id: number, expectedData: string, credential: AuthCredential): boolean { + tryUpdateAuthCredentialIfMatches( + id: number, + expectedData: string, + credential: AuthCredential, + lease?: CredentialRefreshLeaseFence, + ): boolean { const providerStmt = this.#db.prepare("SELECT provider FROM auth_credentials WHERE id = ?"); let providerRow: { provider?: string } | undefined; try { @@ -6109,13 +6169,24 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { const provider = providerRow?.provider ?? ""; const serialized = serializeCredential(provider, credential); if (!serialized) return false; - const result = this.#updateIfMatchesStmt.run( - serialized.credentialType, - serialized.data, - serialized.identityKey, - id, - expectedData, - ) as { changes: number }; + const result = lease + ? (this.#updateIfMatchesWithLeaseStmt.run( + serialized.credentialType, + serialized.data, + serialized.identityKey, + id, + expectedData, + id, + lease.owner, + lease.nowMs, + ) as { changes: number }) + : (this.#updateIfMatchesStmt.run( + serialized.credentialType, + serialized.data, + serialized.identityKey, + id, + expectedData, + ) as { changes: number }); if (result.changes !== 1) return false; if (provider) { this.#purgeSupersededDisabledRows(provider, this.listAuthCredentials(provider)); @@ -6137,10 +6208,24 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore { * the OAuth refresh-failure path to avoid clobbering a peer that rotated the * row between our pre-check and the disable. */ - tryDisableAuthCredentialIfMatches(id: number, expectedData: string, disabledCause: string): boolean { - const result = this.#deleteIfMatchesStmt.run(normalizeDisabledCause(disabledCause), id, expectedData) as { - changes: number; - }; + tryDisableAuthCredentialIfMatches( + id: number, + expectedData: string, + disabledCause: string, + lease?: CredentialRefreshLeaseFence, + ): boolean { + const result = lease + ? (this.#deleteIfMatchesWithLeaseStmt.run( + normalizeDisabledCause(disabledCause), + id, + expectedData, + id, + lease.owner, + lease.nowMs, + ) as { changes: number }) + : (this.#deleteIfMatchesStmt.run(normalizeDisabledCause(disabledCause), id, expectedData) as { + changes: number; + }); return result.changes === 1; } deleteAuthCredentialsForProvider(provider: string, disabledCause: string): void { diff --git a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts index b5179e450..312a073dd 100644 --- a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts +++ b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, setSystemTime, test, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; @@ -36,6 +36,7 @@ describe("AuthStorage OAuth refresh race", () => { afterEach(async () => { vi.restoreAllMocks(); + setSystemTime(); oauthUtils.unregisterOAuthProviders("auth-storage-oauth-refresh-race-test"); store?.close(); store = null; @@ -510,4 +511,110 @@ describe("AuthStorage OAuth refresh race", () => { refresh: "refresh-old", }); }); + + test("does not persist a refresh when durable lease ownership is lost before CAS update", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + + const now = Date.parse("2026-07-10T12:00:00.000Z"); + setSystemTime(new Date(now)); + await authStorage.set("unit-oauth-lease-update", [ + { + type: "oauth", + access: "access-old", + refresh: "refresh-old", + expires: now - 60_000, + }, + ]); + const storedBefore = store.listAuthCredentials("unit-oauth-lease-update"); + expect(storedBefore).toHaveLength(1); + const credentialId = storedBefore[0]!.id; + const stealLease = store.tryAcquireCredentialRefreshLease?.bind(store); + if (!stealLease) throw new Error("test store does not support refresh leases"); + const updateSpy = vi.spyOn(store, "tryUpdateAuthCredentialIfMatches"); + + const result = await authStorage.refreshStoredOAuthCredential("unit-oauth-lease-update", { + credentialFromRow: row => row, + forceRefresh: true, + refresh: async credential => { + // Keep the credential row bytes unchanged while expiring owner A's + // lease. A non-lease-fenced final CAS would still persist this token. + setSystemTime(new Date(now + 16_000)); + expect(stealLease(credentialId, "peer-owner", now + 31_000)).toBe(true); + return { + ...credential, + access: "access-from-lost-owner", + refresh: "refresh-from-lost-owner", + expires: now + 60 * 60_000, + }; + }, + }); + + expect(updateSpy).toHaveBeenCalled(); + expect(result).toMatchObject({ refreshed: false, removed: false }); + expect(result.credential).toMatchObject({ + type: "oauth", + access: "access-old", + refresh: "refresh-old", + }); + const stored = store.listAuthCredentials("unit-oauth-lease-update"); + expect(stored).toHaveLength(1); + expect(stored[0]?.id).toBe(credentialId); + expect(stored[0]?.credential).toMatchObject({ + type: "oauth", + access: "access-old", + refresh: "refresh-old", + }); + }); + + test("does not terminal-disable a credential when durable lease ownership is lost before CAS disable", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + + const now = Date.parse("2026-07-10T12:30:00.000Z"); + setSystemTime(new Date(now)); + await authStorage.set("unit-oauth-lease-disable", [ + { + type: "oauth", + access: "access-old", + refresh: "refresh-old", + expires: now - 60_000, + }, + ]); + const storedBefore = store.listAuthCredentials("unit-oauth-lease-disable"); + expect(storedBefore).toHaveLength(1); + const credentialId = storedBefore[0]!.id; + const stealLease = store.tryAcquireCredentialRefreshLease?.bind(store); + if (!stealLease) throw new Error("test store does not support refresh leases"); + const disableSpy = vi.spyOn(store, "tryDisableAuthCredentialIfMatches"); + + const result = await authStorage.refreshStoredOAuthCredential("unit-oauth-lease-disable", { + credentialFromRow: row => row, + forceRefresh: true, + refresh: async () => { + // The row still contains the same stale refresh token. Only the lease + // fence distinguishes stale owner A from the current row owner. + setSystemTime(new Date(now + 16_000)); + expect(stealLease(credentialId, "peer-owner", now + 31_000)).toBe(true); + throw new Error('HTTP 400 invalid_grant {"error":"invalid_grant"}'); + }, + isDefinitiveFailure: error => error instanceof Error && error.message.includes("invalid_grant"), + disabledCause: error => `oauth refresh failed: ${error instanceof Error ? error.message : String(error)}`, + }); + + expect(disableSpy).toHaveBeenCalled(); + expect(result).toMatchObject({ refreshed: false, removed: false }); + expect(result.credential).toMatchObject({ + type: "oauth", + access: "access-old", + refresh: "refresh-old", + }); + expect(events).toHaveLength(0); + const stored = store.listAuthCredentials("unit-oauth-lease-disable"); + expect(stored).toHaveLength(1); + expect(stored[0]?.id).toBe(credentialId); + expect(stored[0]?.credential).toMatchObject({ + type: "oauth", + access: "access-old", + refresh: "refresh-old", + }); + }); }); diff --git a/packages/coding-agent/src/mcp/manager.ts b/packages/coding-agent/src/mcp/manager.ts index 304a0d245..1cb1adeb0 100644 --- a/packages/coding-agent/src/mcp/manager.ts +++ b/packages/coding-agent/src/mcp/manager.ts @@ -1238,7 +1238,7 @@ export class MCPManager { const material = selectMcpOAuthRefreshMaterial(current, auth); return Boolean(current.refresh && material?.tokenUrl); }, - refresh: current => { + refresh: (current, signal) => { if (current.refresh === REMOTE_REFRESH_SENTINEL) { throw new Error("MCP OAuth refresh token is broker-redacted; local refresh is unavailable"); } @@ -1257,6 +1257,7 @@ export class MCPManager { return refreshMCPOAuthToken(tokenUrl, current.refresh, clientId, clientSecret, resource, { authorizationUrl, stripSameOriginResource: resourceIsFallback, + signal, }); }, mergeRefreshedCredential: (current, refreshed) => { diff --git a/packages/coding-agent/src/mcp/oauth-flow.ts b/packages/coding-agent/src/mcp/oauth-flow.ts index 36e6f9bf7..c68916638 100644 --- a/packages/coding-agent/src/mcp/oauth-flow.ts +++ b/packages/coding-agent/src/mcp/oauth-flow.ts @@ -715,6 +715,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow { */ export interface RefreshMCPOAuthTokenOptions { fetch?: FetchImpl; + signal?: AbortSignal; /** * Authorization-server URL the original grant was minted against. Used to * filter same-origin resource indicators on refresh. Defaults to `tokenUrl`'s @@ -766,6 +767,7 @@ export async function refreshMCPOAuthToken( method: "POST", headers: { "Content-Type": "application/x-www-form-urlencoded" }, body: params.toString(), + signal: optsFromTrailing?.signal, }); if (!response.ok) { diff --git a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts index 8ad061a0f..997e4edad 100644 --- a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts +++ b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts @@ -115,10 +115,11 @@ async function withSharedSQLiteAuth( describe("MCPManager OAuth refresh failure", () => { let manager: MCPManager; let authStorage: AuthStorage; + let store: SqliteAuthCredentialStore; let serverConfig: MCPServerConfig; beforeEach(async () => { - const store = new SqliteAuthCredentialStore(new Database(":memory:")); + store = new SqliteAuthCredentialStore(new Database(":memory:")); authStorage = new AuthStorage(store); await authStorage.reload(); @@ -147,6 +148,7 @@ describe("MCPManager OAuth refresh failure", () => { }); afterEach(() => { + vi.useRealTimers(); authStorage.close(); vi.restoreAllMocks(); }); @@ -169,7 +171,7 @@ describe("MCPManager OAuth refresh failure", () => { undefined, undefined, "https://logfire.example.com/mcp", - { authorizationUrl: undefined, stripSameOriginResource: true }, + { authorizationUrl: undefined, stripSameOriginResource: true, signal: expect.any(AbortSignal) }, ); // The poisoned Bearer must not be re-injected — that is the loop the user // reported (#1908). @@ -221,6 +223,60 @@ describe("MCPManager OAuth refresh failure", () => { const remaining = authStorage.get(CREDENTIAL_ID); expect(remaining).toMatchObject({ type: "oauth", access: "fresh-access", refresh: "fresh-refresh" }); }); + + test("aborts a timed-out token fetch and waits for it before releasing refresh ownership", async () => { + vi.useFakeTimers(); + const fetchCalled = Promise.withResolvers(); + const abortObserved = Promise.withResolvers(); + const allowFetchReject = Promise.withResolvers(); + let capturedSignal: AbortSignal | undefined; + let preparedSettled = false; + const releaseSpy = vi.spyOn(store, "releaseCredentialRefreshLease"); + const fetchImpl = Object.assign( + async (_input: string | URL | Request, init?: RequestInit | BunFetchRequestInit): Promise => { + capturedSignal = init?.signal ?? undefined; + if (!capturedSignal) throw new Error("token refresh fetch did not receive an AbortSignal"); + fetchCalled.resolve(); + capturedSignal.addEventListener( + "abort", + () => { + abortObserved.resolve(); + }, + { once: true }, + ); + await allowFetchReject.promise; + throw capturedSignal.reason ?? new Error("fetch aborted"); + }, + { preconnect: globalThis.fetch.preconnect }, + ); + vi.spyOn(globalThis, "fetch").mockImplementation(fetchImpl); + + const prepared = manager.prepareConfig(serverConfig).finally(() => { + preparedSettled = true; + }); + await fetchCalled.promise; + expect(capturedSignal).toBeDefined(); + + vi.advanceTimersByTime(9_999); + await drainMicrotasks(); + expect(capturedSignal!.aborted).toBe(false); + expect(preparedSettled).toBe(false); + expect(releaseSpy).not.toHaveBeenCalled(); + + vi.advanceTimersByTime(1); + await abortObserved.promise; + expect(capturedSignal!.aborted).toBe(true); + await drainMicrotasks(); + expect(preparedSettled).toBe(false); + expect(releaseSpy).not.toHaveBeenCalled(); + + allowFetchReject.resolve(); + const preparedConfig = await prepared; + + expect(preparedSettled).toBe(true); + expect(releaseSpy).toHaveBeenCalledTimes(1); + expect(getAuthorizationHeader(preparedConfig)).toBe(`Bearer ${STALE_ACCESS}`); + }); }); describe("MCPManager shared SQLite OAuth refresh", () => { From 1b490044ffc26de49a7f33625090b46ba78fe8c1 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:03:53 +0200 Subject: [PATCH 098/205] feat(coding-agent): centralized task orchestration and prompt policy logic - Centralized task concurrency and delegation logic by moving instructions from individual tool descriptions to the system prompt. - Introduced conditional system prompt logic to handle model-specific task policies, including support for GPT-5.6. - Added infrastructure for task concurrency normalization and IRC steering state within the system prompt configuration. - Refactored prompt inputs and session logic to enable dynamic system prompt updates based on model-specific policy cohorts. --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/prompts/system/system-prompt.md | 25 ++++++++--- .../coding-agent/src/prompts/tools/task.md | 9 +--- packages/coding-agent/src/sdk.ts | 6 ++- .../coding-agent/src/session/agent-session.ts | 13 +++--- packages/coding-agent/src/system-prompt.ts | 20 +++++++-- packages/coding-agent/src/task/index.ts | 6 +-- .../coding-agent/src/task/prompt-policy.ts | 8 ++++ .../test/system-prompt-model.test.ts | 45 ++++++++++++++++++- .../coding-agent/test/task/task-spawn.test.ts | 22 --------- 10 files changed, 105 insertions(+), 53 deletions(-) create mode 100644 packages/coding-agent/src/task/prompt-policy.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9ae1ccbd7..2cdfed1f3 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed GPT-5.6 over-delegating work by centralizing task fan-out and concurrency policy in the system prompt: default task mode now uses Codex's explicit-request policy, while eager task mode uses its proactive policy. Other models retain the existing delegation strategy; the task tool keeps only model-independent assignment and coordination guidance. + ## [16.4.1] - 2026-07-10 ### Changed diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 562f7856b..78c1736f5 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -125,7 +125,6 @@ You NEVER open a file hoping. Hope is not a strategy. {{#has tools "grep"}}- Use `{{toolRefs.grep}}` to locate targets.{{/has}} {{#has tools "glob"}}- Use `{{toolRefs.glob}}` to map structure.{{/has}} {{#has tools "read"}}- Use `{{toolRefs.read}}` with offset/limit instead of whole-file reads.{{/has}} -{{#has tools "task"}}- Use `{{toolRefs.task}}` to map unknown code instead of reading file after file yourself.{{/has}} {{#has tools "lsp"}} # LSP @@ -142,9 +141,16 @@ You SHOULD use syntax-aware tools before text hacks: - Use `grep` only for plain-text lookup when structure is irrelevant. {{/ifAny}} -# Delegation -{{#if eagerTasks}} {{#has tools "task"}} +# Delegation +{{#if useCodexTaskPrompt}} +{{#if eagerTasks}} +Proactive multi-agent delegation is active. Any earlier instruction requiring an explicit user request before spawning sub-agents no longer applies. Use sub-agents when parallel work would materially improve speed or quality. This mode remains active until a later multi-agent mode developer message changes it. +{{else}} +Do not spawn sub-agents unless the user or applicable AGENTS.md/skill instructions explicitly ask for sub-agents, delegation, or parallel agent work. +{{/if}} +{{else}} +{{#if eagerTasks}} {{#if eagerTasksAlways}} Delegation is the default here, not the exception. Once the design is settled, you MUST fan the work out to `{{toolRefs.task}}` subagents rather than doing it yourself. Work alone ONLY when one of these is unambiguously true: - A single-file edit under approximately 30 lines @@ -153,8 +159,17 @@ Delegation is the default here, not the exception. Once the design is settled, y Everything else—multi-file changes, refactors, new features, tests, investigations—MUST be decomposed and delegated.{{#if taskBatch}} Batch independent slices into one parallel `{{toolRefs.task}}` call; never serialize what can run concurrently.{{/if}}{{else}}Delegation is preferred here. Once the design is settled, you SHOULD fan substantial work out to `{{toolRefs.task}}` subagents instead of doing everything yourself. Multi-file changes, refactors, new features, tests, and investigations are strong candidates. Use your judgment for small, single-file, or interactive work.{{#if taskBatch}} When you delegate independent slices, batch them into one parallel `{{toolRefs.task}}` call rather than serializing them.{{/if}} {{/if}} -{{/has}} {{/if}} +- Use `{{toolRefs.task}}` to map unknown code instead of reading file after file yourself. +- NEVER abandon phases under scope pressure—delegate, don't shrink. +- Default to parallel for complex changes. Delegate via `{{toolRefs.task}}` for non-importing file edits, multi-subsystem investigation, and decomposable work. +- **Maximize parallelism:** Break work into the widest possible {{#if taskBatch}}array of `tasks[]`{{else}}set of parallel `task` calls{{/if}}. NEVER serialize work that can run concurrently. Tasks touching different files or independent refactors should run in parallel; agents resolve their own file collisions live. +{{#when MAX_CONCURRENCY ">" 0}} +- **Concurrency cap:** At most {{pluralize MAX_CONCURRENCY "subagent" "subagents"}} run at once in this session — anything beyond that just queues, so a {{#if taskBatch}}`tasks[]` batch{{else}}set of parallel `task` calls{{/if}} larger than {{MAX_CONCURRENCY}} only delays results. Keep the fan-out at or under the cap. +{{/when}} +- **Sequence only when necessary:** The only reason to run A before B is if B strictly requires A's output to function (e.g., a core API contract or schema migration). {{#if taskIrcEnabled}}If the missing piece is small, run them in parallel and have B ask A via `irc`!{{/if}} +{{/if}} +{{/has}} EXECUTION WORKFLOW ============== @@ -170,8 +185,6 @@ EXECUTION WORKFLOW # 3. Decompose - Update todos as you go; skip them for trivial requests. Marking a todo done is a transition: start the next in the same turn. -- NEVER abandon phases under scope pressure—delegate, don't shrink. - {{#has tools "task"}}- Default to parallel for complex changes. Delegate via `{{toolRefs.task}}` for non-importing file edits, multi-subsystem investigation, and decomposable work.{{/has}} - Plan only what makes the request work. Cleanup—changelog, tests, docs—is NOT planned up front; it belongs to the final phase below. # 4. Implement diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md index 88d7a3e31..f66b24be8 100644 --- a/packages/coding-agent/src/prompts/tools/task.md +++ b/packages/coding-agent/src/prompts/tools/task.md @@ -2,13 +2,7 @@ Execution does not block your turn: you receive agent and job IDs immediately, and the final results deliver themselves when the subagents finish.{{else}}{{#if batchEnabled}}Run subagents synchronously by passing items in a `tasks[]` batch.{{else}}Run ONE subagent synchronously per call.{{/if}} Execution blocks your turn: the call only returns once the work is completely finished.{{/if}} -# Delegation Strategy -- **Maximize parallelism:** Break work into the widest possible {{#if batchEnabled}}array of `tasks[]`{{else}}set of parallel `task` calls{{/if}}. NEVER serialize work that can run concurrently. Tasks touching different files or independent refactors should run in parallel; agents resolve their own file collisions live. -{{#when MAX_CONCURRENCY ">" 0}} -- **Concurrency cap:** At most {{pluralize MAX_CONCURRENCY "subagent" "subagents"}} run at once in this session — anything beyond that just queues, so a {{#if batchEnabled}}`tasks[]` batch{{else}}set of parallel `task` calls{{/if}} larger than {{MAX_CONCURRENCY}} only delays results. Keep the fan-out at or under the cap. -{{/when}} -- **Sequence only when necessary:** The only reason to run A before B is if B strictly requires A's output to function (e.g., a core API contract or schema migration). {{#if ircEnabled}}If the missing piece is small, run them in parallel and have B ask A via `irc`!{{/if}} -{{#if ircEnabled}}- **Steering delivery:** Parent-to-subagent IRC is delivered immediately as steering; subagents blocked in `job poll` / `irc wait` do not need to poll separately for it.{{/if}} +# Assignment Design - **Role matching:** Assign each subagent a specific `role` (e.g. "Security Reviewer", "DB Migrator"). Do not spawn generic workers. - **No overhead:** Each assignment MUST instruct its agent to skip formatters, linters, and project-wide test suites. You will run those once at the end. - **One-pass agents:** Prefer agents that investigate **and** edit in a single pass; only spin a read-only discovery step (e.g. `scout`) when the affected files are genuinely unknown. @@ -37,6 +31,7 @@ Execution blocks your turn: the call only returns once the work is completely fi # Context and Communication Subagents start blank. They have no access to your conversation history. +{{#if ircEnabled}}- **Steering delivery:** Parent-to-subagent IRC is delivered immediately as steering; subagents blocked in `job poll` / `irc wait` do not need to poll separately for it.{{/if}} {{#if batchEnabled}} - Pass large payloads using `local://` URIs, never inline text. {{else}} diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 8afd72517..4d3fd949a 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -191,6 +191,7 @@ import { import { normalizeToolName, normalizeToolNames } from "./tools/builtin-names"; import { ToolContextStore } from "./tools/context"; import { getImageGenTools } from "./tools/image-gen"; +import { isIrcEnabled } from "./tools/irc"; import { wrapToolWithMetaNotice } from "./tools/output-meta"; import { queueResolveHandler } from "./tools/resolve"; import { ttsTool } from "./tools/tts"; @@ -2444,11 +2445,14 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} eagerTasks, eagerTasksAlways, taskBatch: settings.get("task.batch"), + taskMaxConcurrency: settings.get("task.maxConcurrency"), + taskIrcEnabled: isIrcEnabled(settings, options.taskDepth ?? 0), secretsEnabled, workspaceTree: workspaceTreePromise, includeWorkspaceTree, memoryRootEnabled: memoryBackend.id === "local", - model: settings.get("includeModelInPrompt") ? getActiveModelString() : undefined, + model: getActiveModelString(), + includeModelInPrompt: settings.get("includeModelInPrompt"), personality: agentKind === "sub" ? "none" : settings.get("personality"), renderMermaid: settings.get("tui.renderMermaid"), activeRepoContext, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index e6702ad36..8ec7c69c2 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -286,6 +286,7 @@ import { type SecretObfuscator, } from "../secrets/obfuscator"; import { invalidateHostMetadata } from "../ssh/connection-manager"; +import { usesCodexTaskPrompt } from "../task/prompt-policy"; import { AUTO_THINKING, type ConfiguredThinkingLevel, @@ -6123,19 +6124,17 @@ export class AgentSession { return resolveEditMode(this.#getEditModeSession()); } - /** - * Model key (`provider/id`) currently surfaced in the system prompt, or - * undefined when the model is unset or `includeModelInPrompt` is disabled. - */ + /** Cache key for model-dependent prompt content: displayed id or hidden-policy cohort. */ #currentPromptModelKey(): string | undefined { - if (!this.settings.get("includeModelInPrompt")) return undefined; - return this.model ? formatModelString(this.model) : undefined; + const model = this.model ? formatModelString(this.model) : undefined; + if (!model || this.settings.get("includeModelInPrompt")) return model; + return usesCodexTaskPrompt(model) ? "task-policy:gpt-5.6" : "task-policy:default"; } async #syncAfterModelChange(previousEditMode: EditMode): Promise { const currentEditMode = this.#resolveActiveEditMode(); const editModeChanged = previousEditMode !== currentEditMode && this.getActiveToolNames().includes("edit"); - // The system prompt may surface the active model; a switch makes the cached prompt stale. + // The system prompt selects model-specific policy even when it does not display the model id. const modelChanged = this.#currentPromptModelKey() !== this.#promptModelKey; if (editModeChanged || modelChanged) { await this.refreshBaseSystemPrompt(); diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index f8ddd76ca..7bffe410b 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -22,6 +22,8 @@ import friendlyPersonality from "./prompts/system/personalities/friendly.md" wit import pragmaticPersonality from "./prompts/system/personalities/pragmatic.md" with { type: "text" }; import projectPromptTemplate from "./prompts/system/project-prompt.md" with { type: "text" }; import systemPromptTemplate from "./prompts/system/system-prompt.md" with { type: "text" }; +import { normalizeConcurrencyLimit } from "./task/parallel"; +import { usesCodexTaskPrompt } from "./task/prompt-policy"; import { shortenPath } from "./tools/render-utils"; import { type ActiveRepoContext, resolveActiveRepoContext } from "./utils/active-repo-context"; import { formatLocalCalendarDate } from "./utils/local-date"; @@ -481,8 +483,12 @@ export interface BuildSystemPromptOptions { eagerTasks?: boolean; /** When true, the Eager Tasks section uses the hard MUST/ONLY wording (`task.eager: always`) rather than the softer `preferred` nudge. */ eagerTasksAlways?: boolean; - /** Whether `task.batch` is enabled; gates batch-call guidance in the Eager Tasks section. */ + /** Whether `task.batch` is enabled; selects the centralized delegation guidance's call shape. */ taskBatch?: boolean; + /** Effective task concurrency limit displayed in centralized delegation guidance. Zero means unlimited. */ + taskMaxConcurrency?: number; + /** Whether IRC-backed parallel coordination can be included in delegation policy. */ + taskIrcEnabled?: boolean; /** Rules with alwaysApply=true — their full content is injected into the prompt. */ alwaysApplyRules?: AlwaysApplyRule[]; /** Whether secret obfuscation is active. When true, explains the redaction format in the prompt. */ @@ -491,8 +497,10 @@ export interface BuildSystemPromptOptions { workspaceTree?: WorkspaceTree | Promise; /** Whether the local memory://root summary is active. */ memoryRootEnabled?: boolean; - /** Active model identifier (e.g. "anthropic/claude-opus-4") surfaced to the agent. */ + /** Active model identifier (e.g. "anthropic/claude-opus-4") used by prompt policy and optionally surfaced. */ model?: string; + /** Whether to surface `model` in the workstation block. Model-specific prompt policy still uses it. Default: true. */ + includeModelInPrompt?: boolean; /** Personality preset rendered into the default system prompt. "none" omits the block. Default: "default" */ personality?: Personality; /** Whether to include the workspace directory tree in the system prompt. Default: false */ @@ -536,10 +544,13 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): eagerTasks = false, eagerTasksAlways = false, taskBatch = true, + taskMaxConcurrency = 0, + taskIrcEnabled = false, secretsEnabled = false, workspaceTree: providedWorkspaceTree, memoryRootEnabled = false, model, + includeModelInPrompt = true, personality = "default", includeWorkspaceTree = false, renderMermaid = true, @@ -770,7 +781,8 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): date, dateTime, cwd: promptCwd, - model: model ?? "", + model: includeModelInPrompt ? (model ?? "") : "", + useCodexTaskPrompt: usesCodexTaskPrompt(model), personality: personality === "none" ? "" : PERSONALITY_SPECS[personality].trim(), intentTracing: !!intentField, intentField: intentField ?? "", @@ -780,6 +792,8 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): eagerTasks, eagerTasksAlways, taskBatch, + MAX_CONCURRENCY: normalizeConcurrencyLimit(taskMaxConcurrency), + taskIrcEnabled, secretsEnabled, hasMemoryRoot: memoryRootEnabled, hasObsidian: hasObsidian(), diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 34625c069..6bce9fa08 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -60,7 +60,7 @@ import { } from "./isolation-runner"; import { generateTaskName } from "./name-generator"; import { AgentOutputManager } from "./output-manager"; -import { mapWithConcurrencyLimit, normalizeConcurrencyLimit, Semaphore } from "./parallel"; +import { mapWithConcurrencyLimit, Semaphore } from "./parallel"; import { renderResult, renderCall as renderTaskCall } from "./render"; import { repairTaskParams } from "./repair-args"; import { parseIsolationMode } from "./worktree"; @@ -180,7 +180,6 @@ export function formatResultOutputFallback(result: Pick { return [first, second]; } + function pickTwoModelsWithSameTaskPolicy(): [Model, Model] { + const all = modelRegistry.getAll(); + const first = all[0]; + const second = all.find( + model => + (model.provider !== first.provider || model.id !== first.id) && + usesCodexTaskPrompt(model.id) === usesCodexTaskPrompt(first.id), + ); + if (!first || !second) throw new Error("Expected two distinct models with the same task prompt policy"); + return [first, second]; + } + + function pickModelsAcrossTaskPolicies(): [Model, Model] { + const all = modelRegistry.getAll(); + const defaultPolicy = all.find(model => !usesCodexTaskPrompt(model.id)); + const codexPolicy = all.find(model => usesCodexTaskPrompt(model.id)); + if (!defaultPolicy || !codexPolicy) throw new Error("Expected default-policy and GPT-5.6 models"); + return [defaultPolicy, codexPolicy]; + } + function newSession( model: Model, settings: Settings, @@ -207,8 +228,8 @@ describe("AgentSession model-change prompt refresh", () => { expect(rebuildCount).toBe(1); }); - it("does not rebuild on model change when includeModelInPrompt is disabled", async () => { - const [modelA, modelB] = pickTwoModels(); + it("does not rebuild a hidden-model prompt when the task policy stays the same", async () => { + const [modelA, modelB] = pickTwoModelsWithSameTaskPolicy(); authStorage.setRuntimeApiKey(modelA.provider, "key-a"); authStorage.setRuntimeApiKey(modelB.provider, "key-b"); @@ -226,4 +247,24 @@ describe("AgentSession model-change prompt refresh", () => { expect(rebuildCount).toBe(0); expect(session.agent.state.systemPrompt).toEqual(["initial"]); }); + + it("rebuilds a hidden-model prompt when the task policy changes", async () => { + const [modelA, modelB] = pickModelsAcrossTaskPolicies(); + authStorage.setRuntimeApiKey(modelA.provider, "key-a"); + authStorage.setRuntimeApiKey(modelB.provider, "key-b"); + + let rebuildCount = 0; + session = newSession( + modelA, + Settings.isolated({ "compaction.enabled": false, includeModelInPrompt: false }), + async () => { + rebuildCount++; + return { systemPrompt: ["policy changed"] }; + }, + ); + + await session.setModel(modelB); + expect(rebuildCount).toBe(1); + expect(session.agent.state.systemPrompt).toEqual(["policy changed"]); + }); }); diff --git a/packages/coding-agent/test/task/task-spawn.test.ts b/packages/coding-agent/test/task/task-spawn.test.ts index cb1bac516..21b6f5cb3 100644 --- a/packages/coding-agent/test/task/task-spawn.test.ts +++ b/packages/coding-agent/test/task/task-spawn.test.ts @@ -444,26 +444,4 @@ describe("task spawn routing", () => { gates.get("Fifth")!.resolve(); await Promise.all(jobs.map(job => job.promise)); }); - - it("surfaces task.maxConcurrency in the tool description so the model can self-throttle", async () => { - vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ - agents: [taskAgent], - projectAgentsDir: null, - }); - - const cappedTool = await TaskTool.create(createSession({ settings: { "task.maxConcurrency": 1 } })); - expect(cappedTool.description).toContain("At most 1 subagent"); - expect(cappedTool.description).toContain("Concurrency cap"); - - const fanoutTool = await TaskTool.create(createSession({ settings: { "task.maxConcurrency": 4 } })); - expect(fanoutTool.description).toContain("At most 4 subagents"); - - // `0` = Unlimited in the settings UI; fractional values truncate to 0. - for (const maxConcurrency of [0, 0.5]) { - const unboundedTool = await TaskTool.create( - createSession({ settings: { "task.maxConcurrency": maxConcurrency } }), - ); - expect(unboundedTool.description).not.toContain("Concurrency cap"); - } - }); }); From 7df2ac297e93d4947f267a0a01e3c2396ea0939b Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:08:10 +0200 Subject: [PATCH 099/205] fix(stats): handled legacy session entries with missing cost data - Resolved a crash occurring when syncing legacy session files that lack usage cost breakdowns. - Updated cost resolution to fallback to catalog pricing when stored cost is unavailable or zero. --- packages/stats/CHANGELOG.md | 4 ++++ packages/stats/src/db.ts | 17 ++++++++++++++--- packages/stats/test/sync-serial.test.ts | 16 ++++++++++++++-- 3 files changed, 32 insertions(+), 5 deletions(-) diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index 1904fbbcb..98430028c 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed stats sync crashing on legacy session entries whose usage payload has no cost breakdown; missing costs now use catalog pricing when available. + ## [16.3.9] - 2026-07-06 ### Changed diff --git a/packages/stats/src/db.ts b/packages/stats/src/db.ts index b5ddc2c03..e913989a3 100644 --- a/packages/stats/src/db.ts +++ b/packages/stats/src/db.ts @@ -32,6 +32,14 @@ type ModelCost = { input: number; output: number; cacheRead: number; cacheWrite: type UsageCost = Usage["cost"]; type CostTokens = Pick; +const ZERO_USAGE_COST: UsageCost = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + total: 0, +}; + interface CostBackfillRow { id: number; provider: string; @@ -302,11 +310,14 @@ function calculateCatalogCost(provider: string, modelId: string, tokens: CostTok } function resolveStoredCost(stats: MessageStats): UsageCost { - if (stats.usage.cost.total !== 0) { - return stats.usage.cost; + // `usage.cost` was optional in older session files. Although current + // MessageStats requires it, parsed JSONL can still carry that legacy shape. + const storedCost: UsageCost | undefined = stats.usage.cost; + if (storedCost && storedCost.total !== 0) { + return storedCost; } - return calculateCatalogCost(stats.provider, stats.model, stats.usage) ?? stats.usage.cost; + return calculateCatalogCost(stats.provider, stats.model, stats.usage) ?? storedCost ?? ZERO_USAGE_COST; } function backfillMissingCatalogCosts(database: Database): void { diff --git a/packages/stats/test/sync-serial.test.ts b/packages/stats/test/sync-serial.test.ts index 975045575..f759587e9 100644 --- a/packages/stats/test/sync-serial.test.ts +++ b/packages/stats/test/sync-serial.test.ts @@ -12,11 +12,12 @@ afterEach(() => { vi.restoreAllMocks(); }); -async function writeSessionFile(): Promise { +async function writeSessionFile(options?: { includeCost?: boolean }): Promise { const sessionDir = path.join(getSessionsDir(), "--tmp--sync-serial"); await fs.mkdir(sessionDir, { recursive: true }); const timestamp = new Date().toISOString(); const sessionFile = path.join(sessionDir, "session.jsonl"); + const includeCost = options?.includeCost ?? true; const assistant = { type: "message", id: "assistant-1", @@ -34,7 +35,7 @@ async function writeSessionFile(): Promise { cacheRead: 0, cacheWrite: 0, totalTokens: 3, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + ...(includeCost ? { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } } : {}), }, stopReason: "stop", timestamp: Date.now(), @@ -58,6 +59,17 @@ describe("stats sync serial mode", () => { expect(workerSpy).not.toHaveBeenCalled(); }); + it("syncs legacy session usage without a cost breakdown", async () => { + await writeSessionFile({ includeCost: false }); + + const synced = await syncAllSessions({ workers: 1 }); + const overall = getOverallStats(); + + expect(synced).toEqual({ processed: 1, files: 1 }); + expect(overall.totalRequests).toBe(1); + expect(overall.totalCost).toBeGreaterThan(0); + }); + it("uses the serial parser by default on macOS", async () => { await writeSessionFile(); vi.spyOn(process, "platform", "get").mockReturnValue("darwin"); From 531880c620accc579927d1229968c1f67d8d29b1 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:12:29 +0200 Subject: [PATCH 100/205] feat: improved json serialization for bigint values - Introduced `stringifyJson` helper to preserve bigint precision by serializing them as decimal strings. - Replaced native `JSON.stringify` across compaction and session management modules to prevent serialization errors when handling bigint values in tool arguments. - Added regression tests in `agent` and `coding-agent` packages to ensure bigint tool arguments remain intact through compaction and persistence flows. --- packages/agent/CHANGELOG.md | 4 ++ .../src/compaction/compaction-v2-streaming.ts | 4 +- packages/agent/src/compaction/compaction.ts | 4 +- packages/agent/src/compaction/openai.ts | 8 +-- packages/agent/src/compaction/utils.ts | 4 +- packages/agent/test/remote-compaction.test.ts | 32 ++++++++++ packages/ai/src/dialect/rendering.ts | 3 +- packages/ai/src/providers/openai-shared.ts | 3 +- packages/coding-agent/CHANGELOG.md | 1 + .../src/session/session-manager.ts | 3 +- packages/coding-agent/test/compaction.test.ts | 58 +++++++++++++++++++ packages/utils/CHANGELOG.md | 4 ++ packages/utils/src/json.ts | 13 +++++ 13 files changed, 128 insertions(+), 13 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 568e7f2b0..b45b9b46f 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed serialization of BigInt tool arguments to prevent data loss in remote compaction + ## [16.4.1] - 2026-07-10 ### Fixed diff --git a/packages/agent/src/compaction/compaction-v2-streaming.ts b/packages/agent/src/compaction/compaction-v2-streaming.ts index 8514d4717..e14f6b422 100644 --- a/packages/agent/src/compaction/compaction-v2-streaming.ts +++ b/packages/agent/src/compaction/compaction-v2-streaming.ts @@ -27,7 +27,7 @@ import { OPENAI_HEADER_VALUES, OPENAI_HEADERS, } from "@oh-my-pi/pi-catalog/wire/codex"; -import { $env, logger } from "@oh-my-pi/pi-utils"; +import { $env, logger, stringifyJson } from "@oh-my-pi/pi-utils"; // ============================================================================ // Types & Configuration @@ -329,7 +329,7 @@ async function attemptCompactionV2Streaming( const response = await fetchImpl(endpoint, { method: "POST", headers: buildCompactionV2Headers(model, apiKey, request, codexMetadata), - body: JSON.stringify(body), + body: stringifyJson(body), signal, }); diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 8acf7ffb8..37dd360ca 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -28,7 +28,7 @@ import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses"; import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/providers/openai-shared"; import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking"; -import { logger, prompt } from "@oh-my-pi/pi-utils"; +import { logger, prompt, stringifyJson } from "@oh-my-pi/pi-utils"; import * as snapcompact from "@oh-my-pi/snapcompact"; import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry"; import { ThinkingLevel } from "../thinking"; @@ -406,7 +406,7 @@ export function estimateTokens(message: AgentMessage, options?: { excludeEncrypt } } else if (block.type === "toolCall") { fragments.push(block.name); - fragments.push(JSON.stringify(block.arguments)); + fragments.push(stringifyJson(block.arguments) ?? "null"); } else if (block.type === "redactedThinking") { // Encrypted reasoning blob the provider still bills for on replay; // excluded from the compaction floor for the same reason as above. diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index aeba4603c..7242b8aad 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -43,7 +43,7 @@ import { OPENAI_HEADER_VALUES, OPENAI_HEADERS, } from "@oh-my-pi/pi-catalog/wire/codex"; -import { $env, logger } from "@oh-my-pi/pi-utils"; +import { $env, logger, stringifyJson } from "@oh-my-pi/pi-utils"; export * from "./compaction-v2-streaming"; @@ -419,7 +419,7 @@ export function buildOpenAiNativeHistory( id: itemId, call_id: normalized.callId, name: block.name, - arguments: JSON.stringify(block.arguments), + arguments: stringifyJson(block.arguments) ?? "null", }); } } @@ -549,7 +549,7 @@ export async function requestOpenAiRemoteCompaction( const response = await (opts?.fetch ?? fetch)(endpoint, { method: "POST", headers, - body: JSON.stringify(request), + body: stringifyJson(request), signal: withRequestTimeout(signal, opts?.timeoutMs ?? REMOTE_COMPACTION_TIMEOUT_MS), }); @@ -646,7 +646,7 @@ export async function requestRemoteCompaction( const response = await (opts?.fetch ?? fetch)(endpoint, { method: "POST", headers, - body: JSON.stringify(body), + body: stringifyJson(body), signal: withRequestTimeout(signal, opts?.timeoutMs ?? REMOTE_COMPACTION_TIMEOUT_MS), }); diff --git a/packages/agent/src/compaction/utils.ts b/packages/agent/src/compaction/utils.ts index 43b9df03b..fc3bb6965 100644 --- a/packages/agent/src/compaction/utils.ts +++ b/packages/agent/src/compaction/utils.ts @@ -4,7 +4,7 @@ import type { Message, ToolCall } from "@oh-my-pi/pi-ai"; import { type Dialect, getDialectDefinition } from "@oh-my-pi/pi-ai/dialect"; -import { formatGroupedPaths, prompt } from "@oh-my-pi/pi-utils"; +import { formatGroupedPaths, prompt, stringifyJson } from "@oh-my-pi/pi-utils"; import type { AgentMessage } from "../types"; import fileOperationsTemplate from "./prompts/file-operations.md" with { type: "text" }; import summarizationSystemPrompt from "./prompts/summarization-system.md" with { type: "text" }; @@ -309,7 +309,7 @@ function renderToolCalls(calls: ToolCall[]): string { return calls .map(call => { const argsStr = Object.entries(call.arguments as Record) - .map(([k, v]) => `${k}=${JSON.stringify(v)}`) + .map(([k, v]) => `${k}=${stringifyJson(v) ?? "null"}`) .join(", "); return `${call.name}(${argsStr})`; }) diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index 6051b467b..b1bb2f3b7 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -178,6 +178,38 @@ describe("buildOpenAiNativeHistory custom tool calls", () => { expect(items.find(item => item.type === "function_call")).toBeDefined(); expect(items.find(item => item.type === "custom_tool_call")).toBeUndefined(); }); + + test("preserves bigint tool arguments as exact decimal strings", () => { + const assistant: AssistantMessage = { + role: "assistant", + content: [ + { + type: "toolCall", + id: "call_lookup_1|fc_lookup_1", + name: "lookup", + arguments: { rowId: 9_007_199_254_740_993n }, + }, + ], + timestamp: Date.now(), + provider: "openai", + model: "gpt-5", + api: "openai-responses", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + }; + + const items = buildOpenAiNativeHistory([assistant], makeOpenAiModel()); + const call = items.find(item => item.type === "function_call"); + + expect(call?.arguments).toBe('{"rowId":"9007199254740993"}'); + }); }); const ZERO_USAGE = { diff --git a/packages/ai/src/dialect/rendering.ts b/packages/ai/src/dialect/rendering.ts index 94b69b403..42873bb44 100644 --- a/packages/ai/src/dialect/rendering.ts +++ b/packages/ai/src/dialect/rendering.ts @@ -1,3 +1,4 @@ +import { stringifyJson as stringifyJsonValue } from "@oh-my-pi/pi-utils"; import type { AssistantMessage, Message, ToolCall, ToolResultMessage } from "../types"; import type { DialectRenderOptions, DialectToolResult } from "./types"; @@ -15,7 +16,7 @@ export function harmonyRecipient(name: string): string { } export function stringifyJson(value: unknown): string { - return JSON.stringify(value) ?? "null"; + return stringifyJsonValue(value) ?? "null"; } export function escapeXmlAttr(value: string): string { diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index f70989560..e1b808620 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -27,6 +27,7 @@ import { logger, parseStreamingJson, parseStreamingJsonThrottled, + stringifyJson, structuredCloneJSON, } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; @@ -1604,7 +1605,7 @@ export function convertResponsesAssistantMessage( ...(itemId ? { id: itemId } : {}), call_id: normalized.callId, name: block.name, - arguments: JSON.stringify(block.arguments), + arguments: stringifyJson(block.arguments) ?? "null", }); } diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 48e29d224..159f609b0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed serialization of BigInt values in tool arguments during session compaction - Fixed GPT-5.6 over-delegating work by centralizing task fan-out and concurrency policy in the system prompt: default task mode now uses Codex's explicit-request policy, while eager task mode uses its proactive policy. Other models retain the existing delegation strategy; the task tool keeps only model-independent assignment and coordination guidance. ## [16.4.1] - 2026-07-10 diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index 1f27a1294..be61be344 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -15,6 +15,7 @@ import { getSessionsDir, isEnoent, logger, + stringifyJson, toError, } from "@oh-my-pi/pi-utils"; import { ArtifactManager } from "./artifacts"; @@ -539,7 +540,7 @@ export class SessionManager { } #lineFor(entry: FileEntry): string { - return `${JSON.stringify(prepareEntryForPersistence(entry, this.#blobs))}\n`; + return `${stringifyJson(prepareEntryForPersistence(entry, this.#blobs)) ?? "null"}\n`; } #titleSlotLine(): string { diff --git a/packages/coding-agent/test/compaction.test.ts b/packages/coding-agent/test/compaction.test.ts index 0bcb40263..b7fe41d8d 100644 --- a/packages/coding-agent/test/compaction.test.ts +++ b/packages/coding-agent/test/compaction.test.ts @@ -400,6 +400,64 @@ describe("estimateTokens excludeEncryptedReasoning (compaction floor)", () => { }); }); +describe("bigint tool arguments", () => { + it("preserves exact values through local compaction estimation and summary rendering", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) throw new Error("Expected anthropic/claude-sonnet-4-5 model to exist"); + + const toolCallMessage: AssistantMessage = { + ...createAssistantMessage("", createMockUsage(1_000, 100)), + content: [ + { + type: "toolCall", + id: "call_bigint", + name: "lookup", + arguments: { rowId: 9_007_199_254_740_993n }, + }, + ], + stopReason: "toolUse", + }; + const entries: SessionEntry[] = [ + createMessageEntry(createUserMessage("Look up the row")), + createMessageEntry(toolCallMessage), + createMessageEntry({ + role: "toolResult", + toolCallId: "call_bigint", + toolName: "lookup", + content: [{ type: "text", text: "found" }], + isError: false, + timestamp: Date.now(), + }), + createMessageEntry(createUserMessage("Continue")), + createMessageEntry(createAssistantMessage("Done", createMockUsage(2_000, 100))), + ]; + const preparation = prepareCompaction(entries, { + ...DEFAULT_COMPACTION_SETTINGS, + keepRecentTokens: 1, + remoteEnabled: false, + }); + if (!preparation) throw new Error("Expected compaction preparation"); + + const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(createAssistantMessage("summary")); + const result = await compact(preparation, model, "test-api-key"); + + let renderedPrompts = ""; + for (const call of completeSpy.mock.calls) { + for (const message of call[1].messages) { + if (typeof message.content === "string") { + renderedPrompts += message.content; + continue; + } + for (const block of message.content) { + if (block.type === "text") renderedPrompts += block.text; + } + } + } + expect(renderedPrompts).toContain('"9007199254740993"'); + expect(result.summary).toContain("summary"); + }); +}); + describe("remote compaction setting", () => { it("forwards an explicit initiator override to local summarization requests", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5"); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index fb907b4bd..99d923075 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added stringifyJson utility that supports BigInt serialization + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/utils/src/json.ts b/packages/utils/src/json.ts index 8d7bc7744..e440d2db1 100644 --- a/packages/utils/src/json.ts +++ b/packages/utils/src/json.ts @@ -8,3 +8,16 @@ export function tryParseJson(content: string): T | null { return null; } } + +/** + * Serialize JSON while preserving bigint precision as decimal strings. + * + * Tool arguments normally arrive from JSON providers, but extension hooks and + * host integrations can supply JavaScript bigint values. Native + * `JSON.stringify` throws for those values, which makes otherwise valid agent + * history impossible to persist, replay, or compact. A decimal string is the + * only lossless JSON representation. + */ +export function stringifyJson(value: unknown, space?: string | number): string | undefined { + return JSON.stringify(value, (_key, item) => (typeof item === "bigint" ? item.toString() : item), space); +} From 3e905762288788928fca6fbd6678504120fc0739 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:13:49 +0200 Subject: [PATCH 101/205] test(coding-agent): updated mcp profile auth binding tests - Added AbortSignal expectation to auth binding test assertions. --- packages/coding-agent/test/mcp-profile-auth-binding.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/test/mcp-profile-auth-binding.test.ts b/packages/coding-agent/test/mcp-profile-auth-binding.test.ts index ed19f9dab..e833807dd 100644 --- a/packages/coding-agent/test/mcp-profile-auth-binding.test.ts +++ b/packages/coding-agent/test/mcp-profile-auth-binding.test.ts @@ -202,7 +202,7 @@ describe("per-profile MCP OAuth binding", () => { "embedded-client", "embedded-secret", SERVER_URL, - { authorizationUrl: undefined, stripSameOriginResource: true }, + { authorizationUrl: undefined, stripSameOriginResource: true, signal: expect.any(AbortSignal) }, ); expect(authorizationHeader(prepared)).toBe("Bearer fresh-token"); // Embedded refresh material must survive rotation, or the *next* refresh @@ -315,7 +315,7 @@ describe("per-profile MCP OAuth binding", () => { "my-dcr-client", undefined, SERVER_URL, - { authorizationUrl: undefined, stripSameOriginResource: true }, + { authorizationUrl: undefined, stripSameOriginResource: true, signal: expect.any(AbortSignal) }, ); expect(authorizationHeader(prepared)).toBe("Bearer fresh-token"); }); From a99f46e941616554e7d7180ae18491ffd95dfbaf Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:17:04 +0200 Subject: [PATCH 102/205] chore: update changelogs --- packages/agent/CHANGELOG.md | 2 +- packages/ai/CHANGELOG.md | 8 +++----- packages/catalog/CHANGELOG.md | 7 ++++--- packages/coding-agent/CHANGELOG.md | 6 +++--- packages/stats/CHANGELOG.md | 2 +- packages/utils/CHANGELOG.md | 2 +- 6 files changed, 13 insertions(+), 14 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index b45b9b46f..9e7744081 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed serialization of BigInt tool arguments to prevent data loss in remote compaction +- Fixed serialization of BigInt tool arguments to prevent data loss during remote compaction. ## [16.4.1] - 2026-07-10 diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index fee82347e..7f738f5e8 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,16 +4,15 @@ ### Fixed -- Fixed xAI OAuth Responses continuations replaying OpenAI-only `custom_tool_call`/`custom_tool_call_output` history and `input_image.detail: "original"` frames; replay now downgrades those to xAI-compatible function calls and `detail: "auto"`. ([#5002](https://github.com/can1357/oh-my-pi/issues/5002)) +- Fixed compatibility with xAI by automatically downgrading OpenAI-specific tool calls and image detail settings during message history replays. +- Fixed a race condition in shared SQLite OAuth token refreshes by implementing durable credential ownership and compare-and-set persistence to prevent stale refresh failures. +- Fixed OpenAI Codex requests to include the required version header for newly gated models. ## [16.4.1] - 2026-07-10 ### Changed - Enforced `all_turns` reasoning context for all Responses Lite requests -### Fixed - -- Fixed shared SQLite OAuth refreshes to use durable credential-row ownership plus compare-and-set persistence, preventing stale refresh failures from deleting or overwriting a peer's rotated credential. ([#5081](https://github.com/can1357/oh-my-pi/issues/5081)) ## [16.4.0] - 2026-07-10 @@ -30,7 +29,6 @@ ### Fixed -- Fixed OpenAI Codex turn requests to include the Codex `version` header, matching upstream Codex request metadata for newly gated models. - Fixed xAI SuperGrok multi-account rotation to correctly treat HTTP 403 credit exhaustion and spending limit errors as usage limits, triggering a credential rotation to a sibling account. - Fixed error classification for AWS credential-resolution failures (AwsCredentialsError) to correctly map them as authentication failures. - Fixed OpenAI-compatible chat-completions streams to preserve vLLM-style trailing cached-token usage chunks, ensuring accurate cacheRead and billable input session statistics. diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 3ea5b3229..0ad5cb1bc 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI Codex model discovery to include the Codex version header alongside the client_version query parameter. + ## [16.4.1] - 2026-07-10 ### Added @@ -19,9 +23,6 @@ ### Removed - Removed the generated GPT-5.6 pro-reasoning aliases (`gpt-5.6-{luna,sol,terra}-pro`) from the `openai-codex` subscription provider — pro reasoning is not offered on subscriptions; the `openai` API-key aliases remain -### Fixed - -- Fixed OpenAI Codex model discovery to include the Codex `version` header alongside the `client_version` query parameter. ## [16.4.0] - 2026-07-10 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 159f609b0..8f58b551a 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,8 +4,9 @@ ### Fixed -- Fixed serialization of BigInt values in tool arguments during session compaction -- Fixed GPT-5.6 over-delegating work by centralizing task fan-out and concurrency policy in the system prompt: default task mode now uses Codex's explicit-request policy, while eager task mode uses its proactive policy. Other models retain the existing delegation strategy; the task tool keeps only model-independent assignment and coordination guidance. +- Fixed an issue where BigInt values in tool arguments failed to serialize during session compaction. +- Resolved an issue where GPT-5.6 over-delegated tasks by refining task fan-out and concurrency policies in the system prompt. +- Fixed a race condition in concurrent MCP OAuth token refreshes across processes, ensuring rotating refresh tokens are only refreshed once and preventing stale token errors from clearing valid credentials. ## [16.4.1] - 2026-07-10 @@ -17,7 +18,6 @@ ### Fixed - Fixed MCP OAuth dynamic client registration omitting discovered scopes on the RFC 7591 registration body. Providers such as Clerk bind DCR-created clients to only the scopes declared at registration, then reject the subsequent authorize request when it asks for `openid` (from `scopes_supported`). Registration now includes `config.scopes` when present, matching Claude Code and the scopes already sent on authorize. -- Fixed concurrent MCP OAuth refreshes across OMP processes so rotating refresh tokens are refreshed once, waiters reuse the canonical credential, and stale `invalid_grant` losers cannot clear the winner. ([#5081](https://github.com/can1357/oh-my-pi/issues/5081)) ## [16.4.0] - 2026-07-10 diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index 98430028c..17976f116 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed stats sync crashing on legacy session entries whose usage payload has no cost breakdown; missing costs now use catalog pricing when available. +- Fixed a crash during stats synchronization on legacy session entries that lack a cost breakdown by falling back to catalog pricing when available. ## [16.3.9] - 2026-07-06 diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 99d923075..c71c75024 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added stringifyJson utility that supports BigInt serialization +- Added `stringifyJson` utility with support for BigInt serialization. ## [16.3.12] - 2026-07-08 From fd2d4616a682c5ce4cf1ac7d0dcf88b8ef638aa0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:17:35 +0200 Subject: [PATCH 103/205] chore: bump version to 16.4.2 --- Cargo.lock | 18 +++---- Cargo.toml | 2 +- bun.lock | 69 ++++++++++++++------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 24 +++++----- packages/agent/CHANGELOG.md | 2 + packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 + packages/ai/package.json | 2 +- packages/catalog/CHANGELOG.md | 2 + packages/catalog/package.json | 2 +- packages/catalog/src/models.json | 52 +++++++++++++++++++- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/package.json | 2 +- packages/stats/CHANGELOG.md | 2 + packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/package.json | 2 +- packages/utils/CHANGELOG.md | 2 + packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 27 files changed, 136 insertions(+), 73 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 1bff4d104..049ed97aa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -434,18 +434,18 @@ checksum = "175812e0be2bccb6abe50bb8d566126198344f707e304f45c648fd8f2cc0365e" [[package]] name = "bytemuck" -version = "1.25.0" +version = "1.25.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8efb64bd706a16a1bdde310ae86b351e4d21550d98d056f22f8a7f7a2183fec" +checksum = "d6aedf8ae72766347502cf3cb4f41cf5e9cc37d28bee90f1fdaaae15f9cf9424" dependencies = [ "bytemuck_derive", ] [[package]] name = "bytemuck_derive" -version = "1.10.2" +version = "1.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9abbd1bc6865053c427f7198e6af43bfdedc55ab791faed4fbd361d789575ff" +checksum = "f65693059b6b9c588b9f62fed1cedbf0a8b805631457ea162d68f0de186f3de5" dependencies = [ "proc-macro2", "quote", @@ -2881,7 +2881,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.4.1" +version = "16.4.2" dependencies = [ "anyhow", "ast-grep-core", @@ -2950,7 +2950,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.4.1" +version = "16.4.2" dependencies = [ "async-trait", "libc", @@ -2962,7 +2962,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.4.1" +version = "16.4.2" dependencies = [ "anyhow", "arboard", @@ -3015,7 +3015,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.4.1" +version = "16.4.2" dependencies = [ "anyhow", "brush-builtins", @@ -3064,7 +3064,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.4.1" +version = "16.4.2" dependencies = [ "dashmap", "globset", diff --git a/Cargo.toml b/Cargo.toml index daa3c6881..fc3acc62f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.4.1" +version = "16.4.2" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index 0395e5c66..88776ffbc 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.4.1", + "version": "16.4.2", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.4.1", + "version": "16.4.2", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.4.1", + "version": "16.4.2", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.4.1", + "version": "16.4.2", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.4.1", + "version": "16.4.2", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.4.1", + "version": "16.4.2", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.4.1", + "version": "16.4.2", "devDependencies": { "@types/bun": "catalog:", }, @@ -323,7 +323,6 @@ }, }, "patchedDependencies": { - "@ark/schema@0.56.1": "patches/@ark%2Fschema@0.56.1.patch", "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch", }, "catalog": { @@ -338,18 +337,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.4.1", - "@oh-my-pi/omp-stats": "16.4.1", - "@oh-my-pi/pi-agent-core": "16.4.1", - "@oh-my-pi/pi-ai": "16.4.1", - "@oh-my-pi/pi-catalog": "16.4.1", - "@oh-my-pi/pi-coding-agent": "16.4.1", - "@oh-my-pi/pi-mnemopi": "16.4.1", - "@oh-my-pi/pi-natives": "16.4.1", - "@oh-my-pi/pi-tui": "16.4.1", - "@oh-my-pi/pi-utils": "16.4.1", - "@oh-my-pi/pi-wire": "16.4.1", - "@oh-my-pi/snapcompact": "16.4.1", + "@oh-my-pi/hashline": "16.4.2", + "@oh-my-pi/omp-stats": "16.4.2", + "@oh-my-pi/pi-agent-core": "16.4.2", + "@oh-my-pi/pi-ai": "16.4.2", + "@oh-my-pi/pi-catalog": "16.4.2", + "@oh-my-pi/pi-coding-agent": "16.4.2", + "@oh-my-pi/pi-mnemopi": "16.4.2", + "@oh-my-pi/pi-natives": "16.4.2", + "@oh-my-pi/pi-tui": "16.4.2", + "@oh-my-pi/pi-utils": "16.4.2", + "@oh-my-pi/pi-wire": "16.4.2", + "@oh-my-pi/snapcompact": "16.4.2", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -415,9 +414,9 @@ "@anush008/tokenizers-win32-x64-msvc": ["@anush008/tokenizers-win32-x64-msvc@0.0.0", "", { "os": "win32", "cpu": "x64" }, "sha512-/5kP0G96+Cr6947F0ZetXnmL31YCaN15dbNbh2NHg7TXXRwfqk95+JtPP5Q7v4jbR2xxAmuseBqB4H/V7zKWuw=="], - "@ark/schema": ["@ark/schema@0.56.1", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-1Cf2g9nKD8K/3JGRu+gCCfYw5d4qR8YLLjDs5W5kpmaButCYWAPFUJqSXyBATPjglzCd4tIkp398iPYVs8MjRA=="], + "@ark/schema": ["@ark/schema@0.56.2", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-Qx4D2JFbBWpntiHZaTv7bGG4H/M2rigiknezKg/WVyDSaLdE4YCcWAOoFB7pjjDqHbbV2OqRfntm1nnXvwMexg=="], - "@ark/util": ["@ark/util@0.56.1", "", {}, "sha512-Tp1rTik3q5Z+jAeeDxr5JZpmVIw0miti1ykSEHyZv5Pw3TIJl2xbN6KTacOxITp0l3s9ytlrWd30Zvqcy5vzoQ=="], + "@ark/util": ["@ark/util@0.56.2", "", {}, "sha512-9kU2sUE38FZEGG7l3hamYMBieLYEJh2L1mrYD2eXpT+78EnQSV1bhjxJhnxGBMSTbtwpBSDNSK+K60WvaI/DTQ=="], "@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="], @@ -485,7 +484,7 @@ "@emnapi/core": ["@emnapi/core@1.11.1", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" } }, "sha512-RSvbQmHzdKzNsLYa/wHrbc3KN4sYLKAdPZxqiM2HATqv/SBk2/ENSHpvXGaLOMcsAyz0poEGqkmmKYG3OWiJEQ=="], - "@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], + "@emnapi/runtime": ["@emnapi/runtime@1.11.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-kyOl3X0DuTiT1h2ft8r2fYO8JYtU9a9Xis/zBSiGArNaagCOWx90N1k2wxp18czFDH+OgcWGb5ZP/XMt3dcyPA=="], "@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="], @@ -937,9 +936,9 @@ "argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="], - "arkregex": ["arkregex@0.0.7", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-O/Ltrn9EUSn3ui0KVzfyrWGDUsHlzKxDVBtpQxL/6JmLRMAZAebfSNf/A/J5Ny5S6QIwrXX+RfXsu888HMs35A=="], + "arkregex": ["arkregex@0.0.8", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-PJcx6G1kQTgLKPUbeYlYecDRaKq15AMSGVajlKFYWlPeJRQL+j3dKE6tyMs40HZ99djS1l9Vhl3ezAHy9JBIqQ=="], - "arktype": ["arktype@2.2.2", "", { "dependencies": { "@ark/schema": "0.56.1", "@ark/util": "0.56.1", "arkregex": "0.0.7" } }, "sha512-YYf1xhL2dh5aPZFlsY0RAsxv5HZqfLGLptH2ZP3JidTmsGRW8VOymhPjjMTkerL12vR2YtX0SK4c1mATtae8SA=="], + "arktype": ["arktype@2.2.3", "", { "dependencies": { "@ark/schema": "0.56.2", "@ark/util": "0.56.2", "arkregex": "0.0.8" } }, "sha512-7W+0RLTUNJiBFIIZXwOQxSR8Z273IAd6IvqBeG9+gHnQKFsIx2C0iOtGTmMrPnlX4qLXyc5+ll7A0BIj9WrbTg=="], "async": ["async@3.2.6", "", {}, "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA=="], @@ -1045,7 +1044,7 @@ "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], - "electron-to-chromium": ["electron-to-chromium@1.5.388", "", {}, "sha512-Pl/aJaqOOxYxda3vcx1IKSJimwYXHDkEnGn0F+kG2EE68dDtx2uCinaS+Vih8Z91B9t8CSAbiF/HKyWcnXjhzw=="], + "electron-to-chromium": ["electron-to-chromium@1.5.389", "", {}, "sha512-cEto7aeOqBfU1D+c5py5pE+ooscKE75JifxLBdFUZsqAxRS6y7kebtxAZvICszSl05gPjYHDTjY+lXpyGvpJbg=="], "emnapi": ["emnapi@1.11.2", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-iMt/XQc69fFn2EvcU6tm14HmXKwyy0lnABugsQlqp6xFuZIUuO+ONVSg2mz+MTVF8WbC+bic65AvRXdoldALKg=="], @@ -1493,9 +1492,11 @@ "@opentelemetry/sdk-metrics/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.1", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-RSvbQmHzdKzNsLYa/wHrbc3KN4sYLKAdPZxqiM2HATqv/SBk2/ENSHpvXGaLOMcsAyz0poEGqkmmKYG3OWiJEQ=="], + "@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], - "@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], + "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], + + "@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.2", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-kyOl3X0DuTiT1h2ft8r2fYO8JYtU9a9Xis/zBSiGArNaagCOWx90N1k2wxp18czFDH+OgcWGb5ZP/XMt3dcyPA=="], "@tailwindcss/oxide-wasm32-wasi/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 2709c532d..c1d17c283 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_4_1")] +#[napi(js_name = "__piNativesV16_4_2")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index bfaee536f..f4000b81d 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.4.1", - "@oh-my-pi/omp-stats": "16.4.1", - "@oh-my-pi/pi-agent-core": "16.4.1", - "@oh-my-pi/pi-ai": "16.4.1", - "@oh-my-pi/pi-catalog": "16.4.1", - "@oh-my-pi/pi-coding-agent": "16.4.1", - "@oh-my-pi/pi-mnemopi": "16.4.1", - "@oh-my-pi/pi-natives": "16.4.1", - "@oh-my-pi/pi-tui": "16.4.1", - "@oh-my-pi/pi-utils": "16.4.1", - "@oh-my-pi/pi-wire": "16.4.1", - "@oh-my-pi/snapcompact": "16.4.1", + "@oh-my-pi/hashline": "16.4.2", + "@oh-my-pi/omp-stats": "16.4.2", + "@oh-my-pi/pi-agent-core": "16.4.2", + "@oh-my-pi/pi-ai": "16.4.2", + "@oh-my-pi/pi-catalog": "16.4.2", + "@oh-my-pi/pi-coding-agent": "16.4.2", + "@oh-my-pi/pi-mnemopi": "16.4.2", + "@oh-my-pi/pi-natives": "16.4.2", + "@oh-my-pi/pi-tui": "16.4.2", + "@oh-my-pi/pi-utils": "16.4.2", + "@oh-my-pi/pi-wire": "16.4.2", + "@oh-my-pi/snapcompact": "16.4.2", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 9e7744081..b4b5dd5a8 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.2] - 2026-07-10 + ### Fixed - Fixed serialization of BigInt tool arguments to prevent data loss during remote compaction. diff --git a/packages/agent/package.json b/packages/agent/package.json index 9e561e191..4c8bee08b 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.4.1", + "version": "16.4.2", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7f738f5e8..1a0a75a8d 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.2] - 2026-07-10 + ### Fixed - Fixed compatibility with xAI by automatically downgrading OpenAI-specific tool calls and image detail settings during message history replays. diff --git a/packages/ai/package.json b/packages/ai/package.json index 54033346b..97ea34144 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.4.1", + "version": "16.4.2", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 0ad5cb1bc..5cb50ea54 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.2] - 2026-07-10 + ### Fixed - Fixed OpenAI Codex model discovery to include the Codex version header alongside the client_version query parameter. diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 0ee74b89d..1e169c0fb 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.4.1", + "version": "16.4.2", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 43840585a..35377218c 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -17387,6 +17387,26 @@ "contextWindow": 200000, "maxTokens": 64000 }, + "nemotron-3-ultra-nvfp4": { + "id": "nemotron-3-ultra-nvfp4", + "name": "Nemotron 3 Ultra", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 64000 + }, "swe-1-6": { "id": "swe-1-6", "name": "SWE-1.6", @@ -31094,7 +31114,7 @@ }, "openai/gpt-5.6-terra": { "id": "openai/gpt-5.6-terra", - "name": "GPT-5.6 Terra", + "name": "GPT-5.6 Terra (new)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -62693,6 +62713,36 @@ ] } }, + "gpt-realtime-2.1": { + "id": "gpt-realtime-2.1", + "name": "GPT-Realtime-2.1", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 4, + "output": 24, + "cacheRead": 0.4, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "o1": { "id": "o1", "name": "o1", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8f58b551a..d8041c95c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.2] - 2026-07-10 + ### Fixed - Fixed an issue where BigInt values in tool arguments failed to serialize during session compaction. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index fcf13ea2d..e84b6bac7 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.4.1", + "version": "16.4.2", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 2218fcc40..aa5f96a4b 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.4.1", + "version": "16.4.2", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 8ca2d36d3..241d31d94 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.4.1", + "version": "16.4.2", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 2f377dd9c..b46cfc03e 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -170,7 +170,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_4_1(): void +export declare function __piNativesV16_4_2(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 02486bec8..694a6eb72 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_4_1 = nativeBindings.__piNativesV16_4_1; +export const __piNativesV16_4_2 = nativeBindings.__piNativesV16_4_2; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index a4babd46f..4c401a338 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.4.1", + "version": "16.4.2", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index 2b5292b64..1b3067594 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.4.1", + "version": "16.4.2", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index 17976f116..5897b909c 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.2] - 2026-07-10 + ### Fixed - Fixed a crash during stats synchronization on legacy session entries that lack a cost breakdown by falling back to catalog pricing when available. diff --git a/packages/stats/package.json b/packages/stats/package.json index 0d8293ab6..e180c180b 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.4.1", + "version": "16.4.2", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 49d8bc48a..4e11c20ee 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.4.1", + "version": "16.4.2", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/package.json b/packages/tui/package.json index cd2c1025d..68609a554 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.4.1", + "version": "16.4.2", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index c71c75024..5e14d664b 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.4.2] - 2026-07-10 + ### Added - Added `stringifyJson` utility with support for BigInt serialization. diff --git a/packages/utils/package.json b/packages/utils/package.json index c4be6563f..c1088335d 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.4.1", + "version": "16.4.2", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index 1191a7bd2..ac45b7736 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.4.1", + "version": "16.4.2", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From d048298a353fb694f4c7b42237255f2456fe0390 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:24:18 +0200 Subject: [PATCH 104/205] deps: updated arktype and pin @ark/schema - Updated arktype to version 2.2.2. - Added an override for @ark/schema to version 0.56.1. --- bun.lock | 14 +++++++++----- package.json | 6 ++++-- 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/bun.lock b/bun.lock index 88776ffbc..3d54b7172 100644 --- a/bun.lock +++ b/bun.lock @@ -323,8 +323,12 @@ }, }, "patchedDependencies": { + "@ark/schema@0.56.1": "patches/@ark%2Fschema@0.56.1.patch", "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch", }, + "overrides": { + "@ark/schema": "0.56.1", + }, "catalog": { "@agentclientprotocol/sdk": "0.25.0", "@babel/generator": "^7.29.7", @@ -366,7 +370,7 @@ "@types/turndown": "5.0.6", "@typescript/native-preview": "7.0.0-dev.20260609.1", "@xterm/headless": "^6.0.0", - "arktype": "^2.2.0", + "arktype": "2.2.2", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", @@ -414,9 +418,9 @@ "@anush008/tokenizers-win32-x64-msvc": ["@anush008/tokenizers-win32-x64-msvc@0.0.0", "", { "os": "win32", "cpu": "x64" }, "sha512-/5kP0G96+Cr6947F0ZetXnmL31YCaN15dbNbh2NHg7TXXRwfqk95+JtPP5Q7v4jbR2xxAmuseBqB4H/V7zKWuw=="], - "@ark/schema": ["@ark/schema@0.56.2", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-Qx4D2JFbBWpntiHZaTv7bGG4H/M2rigiknezKg/WVyDSaLdE4YCcWAOoFB7pjjDqHbbV2OqRfntm1nnXvwMexg=="], + "@ark/schema": ["@ark/schema@0.56.1", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-1Cf2g9nKD8K/3JGRu+gCCfYw5d4qR8YLLjDs5W5kpmaButCYWAPFUJqSXyBATPjglzCd4tIkp398iPYVs8MjRA=="], - "@ark/util": ["@ark/util@0.56.2", "", {}, "sha512-9kU2sUE38FZEGG7l3hamYMBieLYEJh2L1mrYD2eXpT+78EnQSV1bhjxJhnxGBMSTbtwpBSDNSK+K60WvaI/DTQ=="], + "@ark/util": ["@ark/util@0.56.1", "", {}, "sha512-Tp1rTik3q5Z+jAeeDxr5JZpmVIw0miti1ykSEHyZv5Pw3TIJl2xbN6KTacOxITp0l3s9ytlrWd30Zvqcy5vzoQ=="], "@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="], @@ -936,9 +940,9 @@ "argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="], - "arkregex": ["arkregex@0.0.8", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-PJcx6G1kQTgLKPUbeYlYecDRaKq15AMSGVajlKFYWlPeJRQL+j3dKE6tyMs40HZ99djS1l9Vhl3ezAHy9JBIqQ=="], + "arkregex": ["arkregex@0.0.7", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-O/Ltrn9EUSn3ui0KVzfyrWGDUsHlzKxDVBtpQxL/6JmLRMAZAebfSNf/A/J5Ny5S6QIwrXX+RfXsu888HMs35A=="], - "arktype": ["arktype@2.2.3", "", { "dependencies": { "@ark/schema": "0.56.2", "@ark/util": "0.56.2", "arkregex": "0.0.8" } }, "sha512-7W+0RLTUNJiBFIIZXwOQxSR8Z273IAd6IvqBeG9+gHnQKFsIx2C0iOtGTmMrPnlX4qLXyc5+ll7A0BIj9WrbTg=="], + "arktype": ["arktype@2.2.2", "", { "dependencies": { "@ark/schema": "0.56.1", "@ark/util": "0.56.1", "arkregex": "0.0.7" } }, "sha512-YYf1xhL2dh5aPZFlsY0RAsxv5HZqfLGLptH2ZP3JidTmsGRW8VOymhPjjMTkerL12vR2YtX0SK4c1mATtae8SA=="], "async": ["async@3.2.6", "", {}, "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA=="], diff --git a/package.json b/package.json index f4000b81d..ecf1df157 100644 --- a/package.json +++ b/package.json @@ -54,7 +54,7 @@ "@types/turndown": "5.0.6", "@typescript/native-preview": "7.0.0-dev.20260609.1", "@xterm/headless": "^6.0.0", - "arktype": "^2.2.0", + "arktype": "2.2.2", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", @@ -92,7 +92,9 @@ "zod": "^4" } }, - "overrides": {}, + "overrides": { + "@ark/schema": "0.56.1" + }, "scripts": { "setup": "bun install && bun run build:native && bun --cwd=packages/coding-agent link && sh scripts/link-omp.sh", "dev": "bun --cwd=packages/coding-agent src/cli.ts", From cf2d4c00b36e6d67e14c2597d379af83f7a22558 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:28:46 +0200 Subject: [PATCH 105/205] fix: stale tests + pin codex version --- .../src/providers/openai-codex-responses.ts | 8 +-- .../auth-storage-block-persistence.test.ts | 4 +- .../ai/test/auth-storage-email-dedupe.test.ts | 10 ++-- packages/ai/test/issue-1701-repro.test.ts | 1 - .../test/openai-codex-responses-lite.test.ts | 3 -- packages/catalog/src/discovery/codex.ts | 51 ++----------------- packages/catalog/src/wire/codex.ts | 5 ++ 7 files changed, 19 insertions(+), 63 deletions(-) diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 67b578d3c..64458dfe9 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -1,9 +1,9 @@ import * as os from "node:os"; import { scheduler } from "node:timers/promises"; -import { resolveCodexClientVersion } from "@oh-my-pi/pi-catalog/discovery/codex"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { CODEX_BASE_URL, + CODEX_CLIENT_VERSION, getCodexAccountId, OPENAI_HEADER_VALUES, OPENAI_HEADERS, @@ -1202,7 +1202,7 @@ async function buildCodexRequestContext( const url = resolveCodexResponsesUrl(baseUrl); const promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); - const codexClientVersion = await resolveCodexClientVersion(undefined, options?.fetch ?? fetch, options?.signal); + const codexClientVersion = CODEX_CLIENT_VERSION; const transformedBody = await buildTransformedCodexRequestBody(model, context, options, promptCacheKey); const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) }; @@ -2502,7 +2502,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses" baseUrl: model.baseUrl || CODEX_BASE_URL, url: "", requestHeaders: {}, - codexClientVersion: packageJson.version, + codexClientVersion: CODEX_CLIENT_VERSION, responsesLite: options?.responsesLite === true, transformedBody: { model: model.id }, rawRequestDump: { @@ -2566,7 +2566,7 @@ export async function prewarmOpenAICodexResponses( transportSessionId ?? crypto.randomUUID(), providerSessionState, ); - const codexClientVersion = await resolveCodexClientVersion(undefined, fetch, options?.signal); + const codexClientVersion = CODEX_CLIENT_VERSION; const requestIdentity = createCodexCompatibilityIdentity(metadataSession); const headers = logger.time( "prewarmCodex:createHeaders", diff --git a/packages/ai/test/auth-storage-block-persistence.test.ts b/packages/ai/test/auth-storage-block-persistence.test.ts index 6488fc10e..c41aad416 100644 --- a/packages/ai/test/auth-storage-block-persistence.test.ts +++ b/packages/ai/test/auth-storage-block-persistence.test.ts @@ -212,7 +212,7 @@ describe("AuthStorage credential block persistence", () => { } }); - it("migrates a v4 auth database to v5 without dropping credential rows", async () => { + it("migrates a v4 auth database to current version 6 without dropping credential rows", async () => { const legacyDb = new Database(dbPath); legacyDb.run(` CREATE TABLE auth_schema_version ( @@ -257,7 +257,7 @@ describe("AuthStorage credential block persistence", () => { const rows = migratedStore.listAuthCredentials(PROVIDER); expect(rows).toHaveLength(1); expect(rows[0]!.credential).toMatchObject({ type: "oauth", access: "legacy-access" }); - expect(readAuthSchemaVersion(dbPath)).toBe(5); + expect(readAuthSchemaVersion(dbPath)).toBe(6); expect(tableExists(dbPath, "auth_credential_blocks")).toBe(true); } finally { migratedStore.close(); diff --git a/packages/ai/test/auth-storage-email-dedupe.test.ts b/packages/ai/test/auth-storage-email-dedupe.test.ts index eb846c3a2..527ec2cd1 100644 --- a/packages/ai/test/auth-storage-email-dedupe.test.ts +++ b/packages/ai/test/auth-storage-email-dedupe.test.ts @@ -431,7 +431,7 @@ describe("AuthStorage openai-codex email dedupe", () => { const freshDbPath = path.join(tempDir, "fresh-schema-agent.db"); const freshStore = await SqliteAuthCredentialStore.open(freshDbPath); try { - expect(readAuthSchemaVersion(freshDbPath)).toBe(5); + expect(readAuthSchemaVersion(freshDbPath)).toBe(6); expect(readTableSql(freshDbPath, "auth_credentials")).not.toContain("unixepoch("); expect(readTableSql(freshDbPath, "auth_credentials")).toContain("strftime('%s','now')"); } finally { @@ -449,7 +449,7 @@ describe("AuthStorage openai-codex email dedupe", () => { id INTEGER PRIMARY KEY CHECK (id = 1), version INTEGER NOT NULL ); - INSERT INTO auth_schema_version(id, version) VALUES (1, 6); + INSERT INTO auth_schema_version(id, version) VALUES (1, 7); CREATE TABLE auth_credentials ( id INTEGER PRIMARY KEY AUTOINCREMENT, provider TEXT NOT NULL, @@ -465,7 +465,7 @@ describe("AuthStorage openai-codex email dedupe", () => { const reopenedStore = await SqliteAuthCredentialStore.open(futureDbPath); try { - expect(readAuthSchemaVersion(futureDbPath)).toBe(6); + expect(readAuthSchemaVersion(futureDbPath)).toBe(7); } finally { reopenedStore.close(); } @@ -491,7 +491,7 @@ describe("AuthStorage openai-codex email dedupe", () => { const reopened = await SqliteAuthCredentialStore.open(reopenDbPath); try { expect(reopened.listAuthCredentials("openai")).toHaveLength(1); - expect(readAuthSchemaVersion(reopenDbPath)).toBe(5); + expect(readAuthSchemaVersion(reopenDbPath)).toBe(6); } finally { reopened.close(); } @@ -547,7 +547,7 @@ describe("AuthStorage openai-codex email dedupe", () => { const migratedStore = await SqliteAuthCredentialStore.open(legacyDbPath); try { - expect(readAuthSchemaVersion(legacyDbPath)).toBe(5); + expect(readAuthSchemaVersion(legacyDbPath)).toBe(6); expect(readTableSql(legacyDbPath, "auth_credentials")).not.toContain("unixepoch("); expect(readTableSql(legacyDbPath, "auth_credentials")).toContain("strftime('%s','now')"); expect(readStoredIdentityRows(legacyDbPath, "openai-codex")).toEqual([ diff --git a/packages/ai/test/issue-1701-repro.test.ts b/packages/ai/test/issue-1701-repro.test.ts index 79318532c..e4373bcc2 100644 --- a/packages/ai/test/issue-1701-repro.test.ts +++ b/packages/ai/test/issue-1701-repro.test.ts @@ -1,6 +1,5 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses"; -import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import type { Context, Model, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types"; diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index a2e16ff7d..a57e515dd 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -100,9 +100,6 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe if (url === "https://api.github.com/repos/openai/codex/releases/latest") { return new Response(JSON.stringify({ tag_name: "rust-v0.0.0" }), { status: 200 }); } - if (url === "https://registry.npmjs.org/@openai%2Fcodex/latest") { - return new Response(JSON.stringify({ version: "0.144.1" }), { status: 200 }); - } if (url.startsWith("https://raw.githubusercontent.com/openai/codex/")) { return new Response("PROMPT", { status: 200, headers: { etag: '"etag"' } }); } diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 118e37fde..33dad7763 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -1,14 +1,11 @@ -import type { FetchImpl } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import type { ModelSpec } from "../types"; -import { discoveryFetch, isRecord } from "../utils"; -import { CODEX_BASE_URL, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../wire/codex"; +import { discoveryFetch } from "../utils"; +import { CODEX_BASE_URL, CODEX_CLIENT_VERSION, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../wire/codex"; const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const; const DEFAULT_CONTEXT_WINDOW = 272_000; const DEFAULT_MAX_TOKENS = 128_000; -const DEFAULT_CODEX_CLIENT_VERSION = "0.99.0"; -const NPM_CODEX_LATEST_URL = "https://registry.npmjs.org/@openai%2Fcodex/latest"; const CODEX_REMOTE_COMPACTION = { enabled: true, api: "openai-codex-responses", @@ -64,8 +61,6 @@ export interface CodexModelDiscoveryOptions { signal?: AbortSignal; /** Optional fetch implementation override for tests. */ fetchFn?: typeof fetch; - /** Optional registry fetch implementation override for client version lookup. */ - registryFetchFn?: typeof fetch; } /** @@ -86,11 +81,7 @@ export async function fetchCodexModels(options: CodexModelDiscoveryOptions): Pro const fetchFn = discoveryFetch(options.fetchFn); const baseUrl = normalizeBaseUrl(options.baseUrl); const paths = normalizePaths(options.paths); - const clientVersion = await resolveCodexClientVersion( - options.clientVersion, - options.registryFetchFn ?? fetchFn, - options.signal, - ); + const clientVersion = normalizeClientVersion(options.clientVersion) ?? CODEX_CLIENT_VERSION; const headers = buildCodexHeaders(options, clientVersion); let sawSuccessfulResponse = false; @@ -169,38 +160,6 @@ function buildCodexHeaders(options: CodexModelDiscoveryOptions, clientVersion: s return headers; } -export async function resolveCodexClientVersion( - clientVersion: string | undefined, - fetchFn: FetchImpl, - signal: AbortSignal | undefined, -): Promise { - const normalizedClientVersion = normalizeClientVersion(clientVersion); - if (normalizedClientVersion) { - return normalizedClientVersion; - } - try { - const response = await fetchFn(NPM_CODEX_LATEST_URL, { - method: "GET", - headers: { Accept: "application/json" }, - signal, - }); - if (!response.ok) { - return DEFAULT_CODEX_CLIENT_VERSION; - } - const payload: unknown = await response.json(); - if (!isRecord(payload)) { - return DEFAULT_CODEX_CLIENT_VERSION; - } - const npmVersion = normalizeClientVersion(payload.version); - return npmVersion ?? DEFAULT_CODEX_CLIENT_VERSION; - } catch (error) { - if (isAbortError(error)) { - throw error; - } - return DEFAULT_CODEX_CLIENT_VERSION; - } -} - function normalizeClientVersion(value: unknown): string | undefined { if (typeof value !== "string") { return undefined; @@ -212,10 +171,6 @@ function normalizeClientVersion(value: unknown): string | undefined { return trimmed; } -function isAbortError(error: unknown): error is Error { - return error instanceof Error && error.name === "AbortError"; -} - function normalizeCodexModels(payload: unknown, baseUrl: string): ModelSpec<"openai-codex-responses">[] | null { const parsedResponse = codexModelsResponseSchema(payload); if (parsedResponse instanceof type.errors) { diff --git a/packages/catalog/src/wire/codex.ts b/packages/catalog/src/wire/codex.ts index 48edeb392..9e0f4d2e0 100644 --- a/packages/catalog/src/wire/codex.ts +++ b/packages/catalog/src/wire/codex.ts @@ -4,6 +4,11 @@ export const CODEX_BASE_URL = "https://chatgpt.com/backend-api"; +/** + * Pinned OpenAI Codex client version (corresponds to @openai/codex package version). + */ +export const CODEX_CLIENT_VERSION = "0.144.1"; + export const OPENAI_HEADERS = { BETA: "OpenAI-Beta", ACCOUNT_ID: "chatgpt-account-id", From e8fd93d2302a6631962f45b1234ab5635c2d6f47 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:31:17 +0200 Subject: [PATCH 106/205] ci: trigger From 219e12b9563b07f300c185a56728d162a040cdce Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:33:39 +0200 Subject: [PATCH 107/205] ditto --- package.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/package.json b/package.json index ecf1df157..1ea076b57 100644 --- a/package.json +++ b/package.json @@ -1,5 +1,5 @@ { - "name": "omp-monorepo", + "name": "omp", "homepage": "https://omp.sh", "private": true, "type": "module", From 7aa1d581c67ad9abb7f2a11b6621da2caf446d54 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 00:37:22 +0200 Subject: [PATCH 108/205] fix(ai): added missing codex import in issue-1701 repro test --- packages/ai/test/issue-1701-repro.test.ts | 1 + 1 file changed, 1 insertion(+) diff --git a/packages/ai/test/issue-1701-repro.test.ts b/packages/ai/test/issue-1701-repro.test.ts index e4373bcc2..79318532c 100644 --- a/packages/ai/test/issue-1701-repro.test.ts +++ b/packages/ai/test/issue-1701-repro.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses"; +import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import type { Context, Model, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types"; From 3188506e6c45d0dbd03731221686659d2d50038e Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 22:39:08 +0000 Subject: [PATCH 109/205] fix(ai): included responses incomplete details Added the required OpenAI Responses incomplete_details field on server-encoded response envelopes and covered completed non-streaming responses. Fixes #5120 --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/providers/openai-responses-server.ts | 10 ++++++++-- packages/ai/test/auth-gateway-openai-responses.test.ts | 2 ++ 3 files changed, 14 insertions(+), 2 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 4dbaf5b00..560f40382 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI Responses server non-streaming envelopes to always include the required `incomplete_details` field, using `null` for completed responses. ([#5120](https://github.com/can1357/oh-my-pi/issues/5120)) + ## [16.4.1] - 2026-07-10 ### Changed diff --git a/packages/ai/src/providers/openai-responses-server.ts b/packages/ai/src/providers/openai-responses-server.ts index e2dd73389..2be91ad60 100644 --- a/packages/ai/src/providers/openai-responses-server.ts +++ b/packages/ai/src/providers/openai-responses-server.ts @@ -568,6 +568,10 @@ function responseStatusForStopReason(message: AssistantMessage): ResponseStatus return "completed"; } +function incompleteDetailsForStatus(status: ResponseStatus): { reason: "max_output_tokens" } | null { + return status === "incomplete" ? { reason: "max_output_tokens" } : null; +} + function buildReasoningItem(part: ThinkingContent): ReasoningOutputItem { const baseId = part.itemId ?? makeReasoningId(); if (part.thinkingSignature) { @@ -718,7 +722,7 @@ function buildResponseEnvelope( model: requestedModelId, output: items, usage, - ...(status === "incomplete" ? { incomplete_details: { reason: "max_output_tokens" } } : {}), + incomplete_details: incompleteDetailsForStatus(status), ...(status === "failed" ? { error: { message: message.errorMessage ?? "response failed" } } : {}), }; } @@ -812,6 +816,7 @@ export function encodeStream( model: requestedModelId, output, usage: null, + incomplete_details: incompleteDetailsForStatus(status), }); const openMessage = (signature?: MessageSignature): OpenMessage => { @@ -1242,7 +1247,7 @@ export function encodeStream( model: requestedModelId, output: items, usage, - ...(status === "incomplete" ? { incomplete_details: { reason: "max_output_tokens" } } : {}), + incomplete_details: incompleteDetailsForStatus(status), ...(status === "failed" ? { error: { message: message?.errorMessage ?? "response failed" } } : {}), @@ -1267,6 +1272,7 @@ export function encodeStream( model: requestedModelId, output: [], error: { message: err instanceof Error ? err.message : String(err) }, + incomplete_details: null, }, }), ), diff --git a/packages/ai/test/auth-gateway-openai-responses.test.ts b/packages/ai/test/auth-gateway-openai-responses.test.ts index cbb68ec66..43f1bbd58 100644 --- a/packages/ai/test/auth-gateway-openai-responses.test.ts +++ b/packages/ai/test/auth-gateway-openai-responses.test.ts @@ -276,6 +276,8 @@ describe("openai-responses encodeResponse", () => { expect(body.object).toBe("response"); expect(body.status).toBe("completed"); + expect(Object.hasOwn(body, "incomplete_details")).toBe(true); + expect(body.incomplete_details).toBeNull(); expect(body.model).toBe("gpt-5-requested"); expect(body.created_at).toBe(1_700_000_000); expect(typeof body.id).toBe("string"); From 851186f5deb2fbc188bba84547c66a0fac0b107d Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 22:41:27 +0000 Subject: [PATCH 110/205] fix(coding-agent): stripped thinking from session titles - Reused the leaked-thinking healer before parsing title markers so visible reasoning envelopes cannot win extraction. - Added regression coverage for and envelopes that contain internal title tags before the real title. Fixes #5122 --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/utils/title-generator.ts | 11 ++++++-- .../coding-agent/test/title-generator.test.ts | 25 +++++++++++++++++++ 3 files changed, 35 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 159f609b0..b95849c2f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,7 @@ - Fixed serialization of BigInt values in tool arguments during session compaction - Fixed GPT-5.6 over-delegating work by centralizing task fan-out and concurrency policy in the system prompt: default task mode now uses Codex's explicit-request policy, while eager task mode uses its proactive policy. Other models retain the existing delegation strategy; the task tool keeps only model-independent assignment and coordination guidance. +- Fixed session title generation including leaked thinking markup from OpenAI-compatible endpoints that ignore reasoning disablement. ([#5122](https://github.com/can1357/oh-my-pi/issues/5122)) ## [16.4.1] - 2026-07-10 diff --git a/packages/coding-agent/src/utils/title-generator.ts b/packages/coding-agent/src/utils/title-generator.ts index a799fa3c0..05ebba553 100644 --- a/packages/coding-agent/src/utils/title-generator.ts +++ b/packages/coding-agent/src/utils/title-generator.ts @@ -4,6 +4,7 @@ import * as path from "node:path"; import { type Api, type AssistantMessage, completeSimple, type Model } from "@oh-my-pi/pi-ai"; +import { StreamMarkupHealing } from "@oh-my-pi/pi-ai/utils/stream-markup-healing"; import { isTerminalHeadless, logger, prompt } from "@oh-my-pi/pi-utils"; import type { ModelRegistry } from "../config/model-registry"; @@ -244,11 +245,17 @@ function extractGeneratedTitle(contentBlocks: AssistantMessage["content"]): stri // Stay lenient: prefer the marker when the model closed it, otherwise // accept a plain sentence after stripping any stray/unclosed tag fragment // (e.g. output truncated before the closing tag). - const marker = TITLE_MARKER_RE.exec(textTitle); - const candidate = marker ? marker[1].trim() : textTitle.replace(/<\/?title>/gi, "").trim(); + const cleanedTextTitle = stripLeakedThinkingMarkup(textTitle); + const marker = TITLE_MARKER_RE.exec(cleanedTextTitle); + const candidate = marker ? marker[1].trim() : cleanedTextTitle.replace(/<\/?title>/gi, "").trim(); return unwrapJsonTitle(candidate); } +function stripLeakedThinkingMarkup(text: string): string { + const healer = new StreamMarkupHealing({ pattern: "thinking" }); + return healer.feed(text) + healer.flushPending(); +} + /** * Unwrap a JSON-shaped response (`{"title": "..."}`, optionally code-fenced) * into the bare title. Models occasionally emit the structured shape they were diff --git a/packages/coding-agent/test/title-generator.test.ts b/packages/coding-agent/test/title-generator.test.ts index c6042c9cc..b17b88fee 100644 --- a/packages/coding-agent/test/title-generator.test.ts +++ b/packages/coding-agent/test/title-generator.test.ts @@ -70,6 +70,31 @@ describe("title generator", () => { expect(options?.disableReasoning).toBe(true); }); + it.each([ + [ + "thinking", + "Thinking process:\nWrong internal scratchpad\n\nFix login button", + ], + [ + "think", + "Thinking process:\nWrong internal scratchpad\n\nFix login button", + ], + ] as const)("ignores leaked <%s> reasoning markup before the visible title", async (_tag, responseText) => { + const model = getModelOrThrow("claude-sonnet-4-5"); + vi.spyOn(ai, "completeSimple").mockResolvedValue({ + stopReason: "stop", + content: [{ type: "text", text: responseText }], + } as never); + + const title = await generateSessionTitle( + "the login button is broken on mobile", + createRegistry(model), + createSettings(model), + ); + + expect(title).toBe("Fix login button"); + }); + it("uses the bundled default prompt when no title prompt file is resolved", async () => { const model = getModelOrThrow("claude-sonnet-4-5"); const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({ From 0420d44d3f9bd9519c77dfaf6a117a4bea0a42db Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 23:00:22 +0000 Subject: [PATCH 111/205] fix(coding-agent): preserved reasoning syntax in titles - Parse only title markers that remain visible after leaked-thinking cleanup so markers inside leaked reasoning are skipped. - Preserve literal reasoning tag syntax inside the chosen title and cover it with a regression test. Fixes #5122 --- .../coding-agent/src/utils/title-generator.ts | 37 +++++++++++++++---- .../coding-agent/test/title-generator.test.ts | 16 ++++++++ 2 files changed, 45 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/utils/title-generator.ts b/packages/coding-agent/src/utils/title-generator.ts index 05ebba553..a4887afd7 100644 --- a/packages/coding-agent/src/utils/title-generator.ts +++ b/packages/coding-agent/src/utils/title-generator.ts @@ -33,7 +33,8 @@ const TERMINAL_TITLE_CONTROL_CHARS = /[\u0000-\u001f\u007f-\u009f]/g; const TITLE_MAX_TOKENS = 1024; /** Matches the title the model wraps in `...`. */ -const TITLE_MARKER_RE = /([\s\S]*?)<\/title>/i; +const TITLE_MARKER_GLOBAL_RE = /<title>([\s\S]*?)<\/title>/gi; +const TITLE_VISIBILITY_SENTINEL = "\uE000omp-title-visible\uE000"; function getTitleModel(registry: ModelRegistry, settings: Settings, currentModel?: Model<Api>): Model<Api> | undefined { const availableModels = registry.getAvailable(); @@ -242,13 +243,33 @@ function extractGeneratedTitle(contentBlocks: AssistantMessage["content"]): stri textTitle += content.text; } } - // Stay lenient: prefer the marker when the model closed it, otherwise - // accept a plain sentence after stripping any stray/unclosed tag fragment - // (e.g. output truncated before the closing tag). - const cleanedTextTitle = stripLeakedThinkingMarkup(textTitle); - const marker = TITLE_MARKER_RE.exec(cleanedTextTitle); - const candidate = marker ? marker[1].trim() : cleanedTextTitle.replace(/<\/?title>/gi, "").trim(); - return unwrapJsonTitle(candidate); + // Stay lenient: prefer the first closed title marker in visible text, then + // fall back to a plain sentence after stripping leaked thinking plus any + // stray/unclosed title tag fragment (e.g. output truncated before closing). + const markedTitle = extractVisibleMarkedTitle(textTitle); + const cleanedTextTitle = + markedTitle ?? + stripLeakedThinkingMarkup(textTitle) + .replace(/<\/?title>/gi, "") + .trim(); + return unwrapJsonTitle(cleanedTextTitle); +} + +function extractVisibleMarkedTitle(text: string): string | undefined { + TITLE_MARKER_GLOBAL_RE.lastIndex = 0; + let marker: RegExpExecArray | null = TITLE_MARKER_GLOBAL_RE.exec(text); + while (marker !== null) { + const title = marker[1]; + if (title !== undefined && isVisibleTitleMarker(text, marker.index)) return title.trim(); + marker = TITLE_MARKER_GLOBAL_RE.exec(text); + } + return undefined; +} + +function isVisibleTitleMarker(text: string, markerIndex: number): boolean { + return stripLeakedThinkingMarkup(`${text.slice(0, markerIndex)}${TITLE_VISIBILITY_SENTINEL}`).endsWith( + TITLE_VISIBILITY_SENTINEL, + ); } function stripLeakedThinkingMarkup(text: string): string { diff --git a/packages/coding-agent/test/title-generator.test.ts b/packages/coding-agent/test/title-generator.test.ts index b17b88fee..cc7cb29f9 100644 --- a/packages/coding-agent/test/title-generator.test.ts +++ b/packages/coding-agent/test/title-generator.test.ts @@ -95,6 +95,22 @@ describe("title generator", () => { expect(title).toBe("Fix login button"); }); + it("preserves in-band reasoning syntax inside the parsed title", async () => { + const model = getModelOrThrow("claude-sonnet-4-5"); + vi.spyOn(ai, "completeSimple").mockResolvedValue({ + stopReason: "stop", + content: [{ type: "text", text: "<title>Fix <think> tag parsing" }], + } as never); + + const title = await generateSessionTitle( + "fix title generation for tag parsing", + createRegistry(model), + createSettings(model), + ); + + expect(title).toBe("Fix tag parsing"); + }); + it("uses the bundled default prompt when no title prompt file is resolved", async () => { const model = getModelOrThrow("claude-sonnet-4-5"); const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({ From 65b0f0532655782f67e26086a4baafafcdcaa3ba Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 10 Jul 2026 23:12:42 +0000 Subject: [PATCH 112/205] fix(coding-agent): preserved markerless thinking titles - Limited markerless fallback cleanup to leading leaked-thinking envelopes so literal reasoning syntax in plain titles survives. - Added markerless regression coverage for think tags and thinking-fence titles. Fixes #5122 --- .../coding-agent/src/utils/title-generator.ts | 18 ++++++++-- .../coding-agent/test/title-generator.test.ts | 33 +++++++++++++++++++ 2 files changed, 48 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/utils/title-generator.ts b/packages/coding-agent/src/utils/title-generator.ts index a4887afd7..69425e79f 100644 --- a/packages/coding-agent/src/utils/title-generator.ts +++ b/packages/coding-agent/src/utils/title-generator.ts @@ -35,6 +35,8 @@ const TITLE_MAX_TOKENS = 1024; /** Matches the title the model wraps in `...`. */ const TITLE_MARKER_GLOBAL_RE = /([\s\S]*?)<\/title>/gi; const TITLE_VISIBILITY_SENTINEL = "\uE000omp-title-visible\uE000"; +const LEADING_THINKING_TAG_RE = /^\s*<(think|thinking|reasoning)>\s*[\s\S]*?<\/\1>\s*/i; +const LEADING_THINKING_FENCE_RE = /^\s*```(?:thinking|reasoning)\b[\s\S]*?```\s*/i; function getTitleModel(registry: ModelRegistry, settings: Settings, currentModel?: Model<Api>): Model<Api> | undefined { const availableModels = registry.getAvailable(); @@ -244,12 +246,12 @@ function extractGeneratedTitle(contentBlocks: AssistantMessage["content"]): stri } } // Stay lenient: prefer the first closed title marker in visible text, then - // fall back to a plain sentence after stripping leaked thinking plus any - // stray/unclosed title tag fragment (e.g. output truncated before closing). + // fall back to a plain sentence after stripping only known leading leaked + // thinking envelopes plus any stray/unclosed title tag fragment. const markedTitle = extractVisibleMarkedTitle(textTitle); const cleanedTextTitle = markedTitle ?? - stripLeakedThinkingMarkup(textTitle) + stripLeadingLeakedThinkingMarkup(textTitle) .replace(/<\/?title>/gi, "") .trim(); return unwrapJsonTitle(cleanedTextTitle); @@ -272,6 +274,16 @@ function isVisibleTitleMarker(text: string, markerIndex: number): boolean { ); } +function stripLeadingLeakedThinkingMarkup(text: string): string { + let current = text; + while (true) { + const withoutTag = current.replace(LEADING_THINKING_TAG_RE, ""); + const withoutFence = withoutTag.replace(LEADING_THINKING_FENCE_RE, ""); + if (withoutFence === current) return current; + current = withoutFence; + } +} + function stripLeakedThinkingMarkup(text: string): string { const healer = new StreamMarkupHealing({ pattern: "thinking" }); return healer.feed(text) + healer.flushPending(); diff --git a/packages/coding-agent/test/title-generator.test.ts b/packages/coding-agent/test/title-generator.test.ts index cc7cb29f9..fdb80cbc4 100644 --- a/packages/coding-agent/test/title-generator.test.ts +++ b/packages/coding-agent/test/title-generator.test.ts @@ -377,6 +377,39 @@ describe("title generator", () => { expect(title).toBe("Fix login button on mobile"); }); + it("preserves a markerless title that mentions a <think> tag", async () => { + const model = getModelFor("deepseek", "deepseek-v4-pro"); + vi.spyOn(ai, "completeSimple").mockResolvedValue({ + stopReason: "stop", + content: [{ type: "text", text: "Fix <think> tag parsing" }], + } as never); + + const title = await generateSessionTitle( + "fix title generation for <think> tag parsing", + createRegistry(model), + createSettings(model), + ); + + expect(title).toBe("Fix <think> tag parsing"); + }); + + it("preserves a markerless title that mentions a ```thinking fence", async () => { + const model = getModelFor("deepseek", "deepseek-v4-pro"); + vi.spyOn(ai, "completeSimple").mockResolvedValue({ + stopReason: "stop", + content: [{ type: "text", text: "Fix ```thinking fence parsing" }], + } as never); + + const title = await generateSessionTitle( + "fix title generation for a ```thinking fence", + createRegistry(model), + createSettings(model), + ); + + expect(title).toContain("```thinking"); + expect(title).toContain("fence"); + }); + it("strips an unclosed <title> tag from a truncated response", async () => { const model = getModelFor("deepseek", "deepseek-v4-pro"); vi.spyOn(ai, "completeSimple").mockResolvedValue({ From a16c60014c89e14362db6bbf54e9732f39f2470c Mon Sep 17 00:00:00 2001 From: roboomp <omp@can.ac> Date: Fri, 10 Jul 2026 23:27:00 +0000 Subject: [PATCH 113/205] fix(coding-agent): hid reasoning envelope title markers - Applied the known reasoning envelope filter when deciding whether a title marker is visible. - Added title extraction regressions for reasoning tag and reasoning fence envelopes before the visible title. Fixes #5122 --- .../coding-agent/src/utils/title-generator.ts | 23 +++++++++++++++++++ .../coding-agent/test/title-generator.test.ts | 14 ++++++++--- 2 files changed, 34 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/utils/title-generator.ts b/packages/coding-agent/src/utils/title-generator.ts index 69425e79f..b8f5850bb 100644 --- a/packages/coding-agent/src/utils/title-generator.ts +++ b/packages/coding-agent/src/utils/title-generator.ts @@ -35,6 +35,8 @@ const TITLE_MAX_TOKENS = 1024; /** Matches the title the model wraps in `<title>...`. */ const TITLE_MARKER_GLOBAL_RE = /([\s\S]*?)<\/title>/gi; const TITLE_VISIBILITY_SENTINEL = "\uE000omp-title-visible\uE000"; +const THINKING_TAG_ENVELOPE_RE = /<(think|thinking|reasoning)>\s*[\s\S]*?<\/\1>/gi; +const THINKING_FENCE_ENVELOPE_RE = /```(?:thinking|reasoning)\b[\s\S]*?```/gi; const LEADING_THINKING_TAG_RE = /^\s*<(think|thinking|reasoning)>\s*[\s\S]*?<\/\1>\s*/i; const LEADING_THINKING_FENCE_RE = /^\s*```(?:thinking|reasoning)\b[\s\S]*?```\s*/i; @@ -269,11 +271,32 @@ function extractVisibleMarkedTitle(text: string): string | undefined { } function isVisibleTitleMarker(text: string, markerIndex: number): boolean { + if (isInsideKnownThinkingEnvelope(text, markerIndex)) return false; return stripLeakedThinkingMarkup(`${text.slice(0, markerIndex)}${TITLE_VISIBILITY_SENTINEL}`).endsWith( TITLE_VISIBILITY_SENTINEL, ); } +function isInsideKnownThinkingEnvelope(text: string, index: number): boolean { + return ( + isInsideEnvelopeMatchedBy(THINKING_TAG_ENVELOPE_RE, text, index) || + isInsideEnvelopeMatchedBy(THINKING_FENCE_ENVELOPE_RE, text, index) + ); +} + +function isInsideEnvelopeMatchedBy(pattern: RegExp, text: string, index: number): boolean { + pattern.lastIndex = 0; + let marker = pattern.exec(text); + while (marker !== null) { + const start = marker.index; + const end = start + marker[0].length; + if (index > start && index < end) return true; + if (start > index) return false; + marker = pattern.exec(text); + } + return false; +} + function stripLeadingLeakedThinkingMarkup(text: string): string { let current = text; while (true) { diff --git a/packages/coding-agent/test/title-generator.test.ts b/packages/coding-agent/test/title-generator.test.ts index fdb80cbc4..b96ffeeb0 100644 --- a/packages/coding-agent/test/title-generator.test.ts +++ b/packages/coding-agent/test/title-generator.test.ts @@ -72,14 +72,22 @@ describe("title generator", () => { it.each([ [ - "thinking", + "<thinking>", "<thinking>Thinking process:\n<title>Wrong internal scratchpad\n\nFix login button", ], [ - "think", + "", "Thinking process:\nWrong internal scratchpad\n\nFix login button", ], - ] as const)("ignores leaked <%s> reasoning markup before the visible title", async (_tag, responseText) => { + [ + "", + "Thinking process:\nWrong internal scratchpad\n\nFix login button", + ], + [ + "```reasoning", + "```reasoning\nThinking process:\nWrong internal scratchpad\n```\nFix login button", + ], + ] as const)("ignores leaked %s reasoning markup before the visible title", async (_marker, responseText) => { const model = getModelOrThrow("claude-sonnet-4-5"); vi.spyOn(ai, "completeSimple").mockResolvedValue({ stopReason: "stop", From 51cc34ac6d28abcd7b1a6b7781e1ebffec5867dd Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 11 Jul 2026 00:04:37 +0000 Subject: [PATCH 114/205] fix(agent): surfaced empty stop retry failures Emitted a failed auto-retry event when empty assistant stop responses exhaust their retry cap so the TUI can show an error instead of silently settling. Fixes #5128 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../coding-agent/src/session/agent-session.ts | 24 +++++++++---------- .../agent-session-empty-stop-guard.test.ts | 22 +++++++++++++++++ 3 files changed, 38 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d8041c95c..2c24bf42c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed empty local-model `stop` responses exhausting their retry cap without surfacing a user-visible retry failure. ([#5128](https://github.com/can1357/oh-my-pi/issues/5128)) + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 8ec7c69c2..dbaa59cea 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -10874,21 +10874,21 @@ export class AgentSession { this.#emptyStopRetryCount++; if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) { - logger.warn("Assistant returned empty stop after retry cap", { - attempts: this.#emptyStopRetryCount - 1, + const attempts = this.#emptyStopRetryCount - 1; + const finalError = "Assistant returned empty stop after retry cap"; + logger.warn(finalError, { + attempts, model: assistantMessage.model, provider: assistantMessage.provider, }); - if (this.#retryAttempt > 0) { - await this.#emitSessionEvent({ - type: "auto_retry_end", - success: false, - attempt: this.#retryAttempt, - finalError: "Assistant returned empty stop after retry cap", - }); - this.#clearPendingRecoveredRetryErrors(); - this.#retryAttempt = 0; - } + await this.#emitSessionEvent({ + type: "auto_retry_end", + success: false, + attempt: this.#retryAttempt > 0 ? this.#retryAttempt : attempts, + finalError, + }); + this.#clearPendingRecoveredRetryErrors(); + this.#retryAttempt = 0; this.#resolveRetry(); // Tool-use orphans corrupt Anthropic message history (tool_result without // matching tool_use). Always remove them even when the retry cap is hit. diff --git a/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts b/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts index 8d3513873..9258161b2 100644 --- a/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts +++ b/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts @@ -270,6 +270,28 @@ describe("AgentSession empty stop guard", () => { expect(emptyAssistantStops(activeBranchMessages)).toHaveLength(1); }); + it("emits failed auto-retry end when repeated empty stops exhaust the retry cap", async () => { + const { session, mock } = await createHarness([emptyStop(), emptyStop(), emptyStop(), emptyStop()]); + const retryEndEvents: Array> = []; + session.subscribe(event => { + if (event.type === "auto_retry_end") { + retryEndEvents.push(event); + } + }); + + await expectPromptCompletes(session.prompt("answer without tools")); + await session.waitForIdle(); + + expect(mock.calls).toHaveLength(4); + expect(retryEndEvents).toHaveLength(1); + expect(retryEndEvents[0]).toMatchObject({ + type: "auto_retry_end", + success: false, + attempt: 3, + }); + expect(retryEndEvents[0]?.finalError).toContain("empty stop"); + }); + it("ends auto-retry state when empty stop retries hit the cap", async () => { vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); const { session, mock } = await createHarness( From 159484ca6ffcdf3813ba64987dce2d7eebfc3175 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 11 Jul 2026 00:58:34 +0000 Subject: [PATCH 115/205] fix(commit): created commits before agent teardown - Ran commit host completion before commit-agent session disposal so mnemopi/autolearn teardown cannot preempt a valid proposal. - Converted missing commit-agent host outputs and split-plan gaps into thrown errors so omp commit cannot resolve into exit 0 without creating a commit. - Preserved caller GPG_TTY state instead of forcing a bogus signing TTY in git and non-interactive subprocess environments. Fixes #4794 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/commit/agentic/agent.ts | 4 ++ .../coding-agent/src/commit/agentic/index.ts | 67 +++++++++++++------ .../src/exec/non-interactive-env.ts | 1 - packages/coding-agent/src/utils/git.ts | 1 - .../test/commit-agentic-attribution.test.ts | 46 +++++++++++++ .../test/commit-command-exit.test.ts | 20 ++++++ .../test/git-process-config.test.ts | 38 +++++++++++ .../test/non-interactive-env.test.ts | 12 ++++ 9 files changed, 170 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..063d52e2e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed `omp commit` agent sessions so a valid proposal is committed before session teardown can dispose mnemopi/autolearn resources, missing required host outputs now fail non-zero instead of returning cleanly, and git subprocesses no longer force `GPG_TTY=not a tty` on signing-enabled repositories ([#4794](https://github.com/can1357/oh-my-pi/issues/4794)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/commit/agentic/agent.ts b/packages/coding-agent/src/commit/agentic/agent.ts index 68b529e97..98b06540d 100644 --- a/packages/coding-agent/src/commit/agentic/agent.ts +++ b/packages/coding-agent/src/commit/agentic/agent.ts @@ -29,6 +29,7 @@ export interface CommitAgentInput { requireChangelog: boolean; diffText?: string; existingChangelogEntries?: ExistingChangelogEntries[]; + onComplete?: (state: CommitAgentState) => Promise | void; } export interface ExistingChangelogEntries { @@ -175,6 +176,9 @@ export async function runCommitAgentSession(input: CommitAgentInput): Promise { } process.stdout.write("● Starting commit agent...\n"); - let commitState: CommitAgentState; - let usedFallback = false; + let agentSessionCompleted = false; try { - commitState = await runCommitAgentSession({ + await runCommitAgentSession({ cwd, model: agentModel, thinkingLevel: agentThinkingLevel, @@ -139,42 +138,67 @@ export async function runAgenticCommit(args: CommitCommandArgs): Promise { requireChangelog: !args.noChangelog && changelogTargets.length > 0, diffText: diff, existingChangelogEntries, + onComplete: async commitState => { + agentSessionCompleted = true; + await completeAgentCommitState(commitState, { + cwd, + dryRun: args.dryRun, + push: args.push, + noChangelog: args.noChangelog, + changelogTargets, + numstat, + }); + }, }); + return; } catch (error) { + if (agentSessionCompleted) { + throw error; + } const errorMessage = error instanceof Error ? error.message : String(error); process.stderr.write(`Agent error: ${errorMessage}\n`); if (error instanceof Error && error.stack && $env.DEBUG) { process.stderr.write(`${error.stack}\n`); } process.stdout.write("● Using fallback commit generation...\n"); - commitState = { proposal: generateFallbackProposal(numstat) }; - usedFallback = true; + const fallbackProposal = generateFallbackProposal(numstat); + await runSingleCommit(fallbackProposal, { cwd, dryRun: args.dryRun, push: args.push }); + return; } +} - if (!usedFallback && !commitState.proposal && !commitState.splitProposal) { +async function completeAgentCommitState( + commitState: CommitAgentState, + ctx: CommitExecutionContext & { + noChangelog: boolean; + changelogTargets: string[]; + numstat: NumstatEntry[]; + }, +): Promise { + let usedFallback = false; + if (!commitState.proposal && !commitState.splitProposal) { if ($env.PI_COMMIT_NO_FALLBACK?.toLowerCase() !== "true") { process.stdout.write("● Agent did not provide proposal, using fallback...\n"); - commitState.proposal = generateFallbackProposal(numstat); + commitState.proposal = generateFallbackProposal(ctx.numstat); usedFallback = true; } } let updatedChangelogFiles: string[] = []; - if (!args.noChangelog && changelogTargets.length > 0 && !usedFallback) { + if (!ctx.noChangelog && ctx.changelogTargets.length > 0 && !usedFallback) { if (!commitState.changelogProposal) { - process.stderr.write("Commit agent did not provide changelog entries.\n"); - return; + throw new Error("Commit agent did not provide changelog entries."); } process.stdout.write("● Applying changelog entries...\n"); const updated = await applyChangelogProposals({ - cwd, + cwd: ctx.cwd, proposals: commitState.changelogProposal.entries, - dryRun: args.dryRun, + dryRun: ctx.dryRun, onProgress: message => { process.stdout.write(` ├─ ${message}\n`); }, }); - updatedChangelogFiles = updated.map(filePath => path.relative(cwd, filePath)); + updatedChangelogFiles = updated.map(filePath => path.relative(ctx.cwd, filePath)); if (updated.length > 0) { for (const filePath of updated) { process.stdout.write(` └─ ${filePath}\n`); @@ -185,21 +209,21 @@ export async function runAgenticCommit(args: CommitCommandArgs): Promise { } if (commitState.proposal) { - await runSingleCommit(commitState.proposal, { cwd, dryRun: args.dryRun, push: args.push }); + await runSingleCommit(commitState.proposal, ctx); return; } if (commitState.splitProposal) { await runSplitCommit(commitState.splitProposal, { - cwd, - dryRun: args.dryRun, - push: args.push, + cwd: ctx.cwd, + dryRun: ctx.dryRun, + push: ctx.push, additionalFiles: updatedChangelogFiles, }); return; } - process.stderr.write("Commit agent did not provide a proposal.\n"); + throw new Error("Commit agent did not provide a proposal."); } async function runSingleCommit(proposal: CommitProposal, ctx: CommitExecutionContext): Promise { @@ -212,6 +236,7 @@ async function runSingleCommit(proposal: CommitProposal, ctx: CommitExecutionCon process.stdout.write(`${commitMessage}\n`); return; } + process.stdout.write("● Creating commit...\n"); await git.commit(ctx.cwd, commitMessage); process.stdout.write("Commit created.\n"); if (ctx.push) { @@ -235,8 +260,7 @@ async function runSplitCommit( const plannedFiles = new Set(plan.commits.flatMap(commit => commit.changes.map(change => change.path))); const missingFiles = stagedFiles.filter(file => !plannedFiles.has(file)); if (missingFiles.length > 0) { - process.stderr.write(`Split commit plan missing staged files: ${missingFiles.join(", ")}\n`); - return; + throw new Error(`Split commit plan missing staged files: ${missingFiles.join(", ")}`); } if (ctx.dryRun) { @@ -268,6 +292,7 @@ async function runSplitCommit( throw new Error(order.error); } + process.stdout.write("● Creating split commits...\n"); const stagedDiff = await git.diff(ctx.cwd, { cached: true, binary: true }); await git.stage.reset(ctx.cwd); for (const commitIndex of order) { diff --git a/packages/coding-agent/src/exec/non-interactive-env.ts b/packages/coding-agent/src/exec/non-interactive-env.ts index c1b58ba91..1d25df10e 100644 --- a/packages/coding-agent/src/exec/non-interactive-env.ts +++ b/packages/coding-agent/src/exec/non-interactive-env.ts @@ -15,7 +15,6 @@ export const NON_INTERACTIVE_ENV: Readonly> = { LESS: "FRX", // Disable terminal features that can block the process. TERM: "dumb", - GPG_TTY: "not a tty", NO_COLOR: "1", PYTHONUNBUFFERED: "1", // Disable editor and terminal credential prompts. diff --git a/packages/coding-agent/src/utils/git.ts b/packages/coding-agent/src/utils/git.ts index d0a9eff5b..2974b0c68 100644 --- a/packages/coding-agent/src/utils/git.ts +++ b/packages/coding-agent/src/utils/git.ts @@ -197,7 +197,6 @@ const GIT_NON_INTERACTIVE_ENV = { GIT_ASKPASS: "true", GIT_EDITOR: "true", GIT_TERMINAL_PROMPT: "0", - GPG_TTY: "not a tty", SSH_ASKPASS: "/usr/bin/false", } satisfies Record; const GH_NON_INTERACTIVE_ENV = { diff --git a/packages/coding-agent/test/commit-agentic-attribution.test.ts b/packages/coding-agent/test/commit-agentic-attribution.test.ts index 54bf677fb..f3ee64dc0 100644 --- a/packages/coding-agent/test/commit-agentic-attribution.test.ts +++ b/packages/coding-agent/test/commit-agentic-attribution.test.ts @@ -46,4 +46,50 @@ describe("commit agent prompt attribution", () => { expect(prompt.options?.expandPromptTemplates).toBe(false); } }); + + it("runs completion before session disposal", async () => { + const events: string[] = []; + const session = { + prompt: async () => {}, + subscribe: () => () => {}, + dispose: async () => { + events.push("dispose"); + }, + }; + + vi.spyOn(sdkModule, "createAgentSession").mockResolvedValue({ session } as unknown as CreateAgentSessionResult); + vi.spyOn(toolsModule, "createCommitTools").mockImplementation(options => { + options.state.proposal = { + analysis: { + type: "fix", + scope: "commit", + details: [], + issueRefs: [], + }, + summary: "create commit before teardown", + warnings: [], + }; + return []; + }); + + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) { + throw new Error("Expected claude-sonnet-4-5 model to exist"); + } + + await runCommitAgentSession({ + cwd: "/tmp", + model, + settings: Settings.isolated(), + modelRegistry: {} as never, + authStorage: {} as never, + changelogTargets: [], + requireChangelog: false, + onComplete: state => { + events.push(state.proposal?.summary ?? "missing proposal"); + }, + }); + + expect(events).toEqual(["create commit before teardown", "dispose"]); + }); }); diff --git a/packages/coding-agent/test/commit-command-exit.test.ts b/packages/coding-agent/test/commit-command-exit.test.ts index f75247049..f5cd985c6 100644 --- a/packages/coding-agent/test/commit-command-exit.test.ts +++ b/packages/coding-agent/test/commit-command-exit.test.ts @@ -32,4 +32,24 @@ describe("omp commit command lifecycle (issue #1041)", () => { expect(runCommitSpy.mock.invocationCallOrder[0]).toBeLessThan(quitSpy.mock.invocationCallOrder[0]); expect(quitSpy).toHaveBeenCalledWith(0); }); + + it("does not convert commit pipeline failures into exit 0", async () => { + const initThemeSpy = vi.spyOn(themeModule, "initTheme").mockResolvedValue(undefined); + const runCommitSpy = vi + .spyOn(commitModule, "runCommitCommand") + .mockRejectedValue(new Error("commit was not created")); + const quitSpy = vi.spyOn(postmortem, "quit").mockResolvedValue(undefined); + + const command = new CommitCommand([], { + bin: "omp", + version: "0.0.0-test", + commands: new Map(), + }); + + await expect(command.run()).rejects.toThrow("commit was not created"); + + expect(initThemeSpy).toHaveBeenCalledTimes(1); + expect(runCommitSpy).toHaveBeenCalledTimes(1); + expect(quitSpy).not.toHaveBeenCalled(); + }); }); diff --git a/packages/coding-agent/test/git-process-config.test.ts b/packages/coding-agent/test/git-process-config.test.ts index 65e5ee039..68f508f15 100644 --- a/packages/coding-agent/test/git-process-config.test.ts +++ b/packages/coding-agent/test/git-process-config.test.ts @@ -110,4 +110,42 @@ describe("git subprocess config", () => { "HEAD:refs/heads/feature", ]); }); + + it("preserves the caller's GPG_TTY for signing-capable commands", async () => { + const originalGpgTty = process.env.GPG_TTY; + const spawnCalls: SpawnCall[] = []; + vi.spyOn(Bun, "spawn").mockImplementation(createSpawnMock(spawnCalls)); + + process.env.GPG_TTY = "/dev/pts/42"; + try { + await git.commit("/work/pi", "fix: preserve signing tty"); + } finally { + if (originalGpgTty === undefined) { + delete process.env.GPG_TTY; + } else { + process.env.GPG_TTY = originalGpgTty; + } + } + + expect(spawnCalls).toHaveLength(1); + expect(spawnCalls[0]?.options.env?.GPG_TTY).toBe("/dev/pts/42"); + }); + + it("does not invent a bogus GPG_TTY when the caller has none", async () => { + const originalGpgTty = process.env.GPG_TTY; + const spawnCalls: SpawnCall[] = []; + vi.spyOn(Bun, "spawn").mockImplementation(createSpawnMock(spawnCalls)); + + delete process.env.GPG_TTY; + try { + await git.commit("/work/pi", "fix: allow gui pinentry"); + } finally { + if (originalGpgTty !== undefined) { + process.env.GPG_TTY = originalGpgTty; + } + } + + expect(spawnCalls).toHaveLength(1); + expect(spawnCalls[0]?.options.env).not.toHaveProperty("GPG_TTY"); + }); }); diff --git a/packages/coding-agent/test/non-interactive-env.test.ts b/packages/coding-agent/test/non-interactive-env.test.ts index 57c4ca878..0eec8035b 100644 --- a/packages/coding-agent/test/non-interactive-env.test.ts +++ b/packages/coding-agent/test/non-interactive-env.test.ts @@ -44,4 +44,16 @@ describe("buildNonInteractiveEnv", () => { expect(env.LANG).toBeUndefined(); expect(env.LC_ALL).toBeUndefined(); }); + + it("does not invent a bogus GPG_TTY", () => { + const env = buildNonInteractiveEnv(undefined, {}, "linux"); + + expect(env).not.toHaveProperty("GPG_TTY"); + }); + + it("preserves per-command GPG_TTY overrides", () => { + const env = buildNonInteractiveEnv({ GPG_TTY: "/dev/pts/7" }, {}, "linux"); + + expect(env.GPG_TTY).toBe("/dev/pts/7"); + }); }); From f534112957be6c6d0b1b7e6647c16a953b938bfb Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 11 Jul 2026 02:28:27 +0000 Subject: [PATCH 116/205] fix(coding-agent): bound startup changelog rendering - Treated missing or invalid changelog markers as first install and persisted the current version without replaying historical notes. - Shared bounded changelog rendering between startup and recent changelog views, with a 64 KiB startup cap and full-history hint on truncation. - Added marker, truncation, recent/full rendering, and PTY startup regression coverage. Fixes #5135 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/main.ts | 26 +- .../modes/controllers/command-controller.ts | 18 +- .../src/slash-commands/builtin-registry.ts | 16 +- packages/coding-agent/src/utils/changelog.ts | 116 ++++++++- .../coding-agent/test/utils/changelog.test.ts | 232 ++++++++++++++++++ 6 files changed, 372 insertions(+), 40 deletions(-) create mode 100644 packages/coding-agent/test/utils/changelog.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d8041c95c..025300fc8 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed first-run interactive startup rendering the full packaged changelog when the last-seen marker is missing, malformed, or unreadable. Startup upgrade notes now show at most three unseen releases and cap Markdown source at 64 KiB; `/changelog full` remains the explicit full-history path. ([#5135](https://github.com/can1357/oh-my-pi/issues/5135)) + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index ee25259b0..0b2b119d9 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -78,9 +78,10 @@ import { concreteThinkingLevel, parseConfiguredThinkingLevel } from "./thinking" import type { LspStartupServerInfo } from "./tools"; import { getChangelogPath, - getNewEntries, parseChangelog, + parseChangelogVersion, readLastChangelogVersion, + selectStartupChangelog, writeLastChangelogVersion, } from "./utils/changelog"; import { EventBus } from "./utils/event-bus"; @@ -610,6 +611,11 @@ async function getChangelogForDisplay(parsed: Args): Promise } const lastVersion = await readLastChangelogVersion(); + const parsedLastVersion = parseChangelogVersion(lastVersion); + if (!parsedLastVersion) { + await writeLastChangelogVersion(VERSION); + return undefined; + } if (lastVersion === VERSION) { // Steady state: user already saw the current version's changelog. Skip the file read + parse. return undefined; @@ -617,18 +623,12 @@ async function getChangelogForDisplay(parsed: Args): Promise const changelogPath = getChangelogPath(); const entries = await parseChangelog(changelogPath); - - if (!lastVersion) { - if (entries.length > 0) { - await writeLastChangelogVersion(VERSION); - return entries.map(e => e.content).join("\n\n"); - } - } else { - const newEntries = getNewEntries(entries, lastVersion); - if (newEntries.length > 0) { - await writeLastChangelogVersion(VERSION); - return newEntries.map(e => e.content).join("\n\n"); - } + const startupChangelog = selectStartupChangelog(entries, lastVersion, VERSION); + if (startupChangelog.persistCurrentVersion) { + await writeLastChangelogVersion(VERSION); + } + if (startupChangelog.markdown) { + return startupChangelog.markdown; } return undefined; diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index b6573e178..ef6363d89 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -47,7 +47,12 @@ import { limitMatchesActiveAccount } from "../../slash-commands/helpers/active-o import { outputMeta } from "../../tools/output-meta"; import { resolveToCwd, stripOuterDoubleQuotes } from "../../tools/path-utils"; import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; -import { getChangelogPath, parseChangelog } from "../../utils/changelog"; +import { + getChangelogPath, + parseChangelog, + RECENT_CHANGELOG_ENTRY_LIMIT, + renderChangelogEntries, +} from "../../utils/changelog"; import { copyToClipboard } from "../../utils/clipboard"; import { openPath } from "../../utils/open"; import { setSessionTerminalTitle } from "../../utils/title-generator"; @@ -487,16 +492,9 @@ export class CommandController { async handleChangelogCommand(showFull = false): Promise { const changelogPath = getChangelogPath(); const allEntries = await parseChangelog(changelogPath); - // Default to showing only the latest 3 versions unless --full is specified - // allEntries comes from parseChangelog with newest first, reverse to show oldest->newest - const entriesToShow = showFull ? allEntries : allEntries.slice(0, 3); + const entriesToShow = showFull ? allEntries : allEntries.slice(0, RECENT_CHANGELOG_ENTRY_LIMIT); const changelogMarkdown = - entriesToShow.length > 0 - ? [...entriesToShow] - .reverse() - .map(e => e.content) - .join("\n\n") - : "No changelog entries found."; + entriesToShow.length > 0 ? renderChangelogEntries(entriesToShow).markdown : "No changelog entries found."; const title = showFull ? "Full Changelog" : "Recent Changes"; const hint = showFull ? "" diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 5f9bdfc2e..0377624a9 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -34,7 +34,12 @@ import { resolveResumableSession } from "../session/session-listing"; import { formatShakeSummary, type ShakeMode } from "../session/shake-types"; import { expandTilde, resolveToCwd } from "../tools/path-utils"; import { urlHyperlinkAlways } from "../tui"; -import { getChangelogPath, parseChangelog } from "../utils/changelog"; +import { + getChangelogPath, + parseChangelog, + RECENT_CHANGELOG_ENTRY_LIMIT, + renderChangelogEntries, +} from "../utils/changelog"; import { copyToClipboard } from "../utils/clipboard"; import { CollabQrCodeComponent } from "./helpers/collab-qrcode"; import { buildContextReportText } from "./helpers/context-report"; @@ -1082,17 +1087,12 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ const changelogPath = getChangelogPath(); const allEntries = await parseChangelog(changelogPath); const showFull = command.args.trim().toLowerCase() === "full"; - const entriesToShow = showFull ? allEntries : allEntries.slice(0, 3); + const entriesToShow = showFull ? allEntries : allEntries.slice(0, RECENT_CHANGELOG_ENTRY_LIMIT); if (entriesToShow.length === 0) { await runtime.output("No changelog entries found."); return commandConsumed(); } - await runtime.output( - [...entriesToShow] - .reverse() - .map(entry => entry.content) - .join("\n\n"), - ); + await runtime.output(renderChangelogEntries(entriesToShow).markdown); return commandConsumed(); }, handleTui: async (command, runtime) => { diff --git a/packages/coding-agent/src/utils/changelog.ts b/packages/coding-agent/src/utils/changelog.ts index ac12bb401..fcc12be85 100644 --- a/packages/coding-agent/src/utils/changelog.ts +++ b/packages/coding-agent/src/utils/changelog.ts @@ -7,6 +7,27 @@ export interface ChangelogEntry { content: string; } +/** Number of changelog releases shown by automatic and default recent views. */ +export const RECENT_CHANGELOG_ENTRY_LIMIT = 3; +/** Maximum Markdown source bytes allowed in automatic startup release notes. */ +export const STARTUP_CHANGELOG_MAX_BYTES = 64 * 1024; +/** Hint appended when automatic startup release notes are truncated. */ +export const STARTUP_CHANGELOG_FULL_HINT = "Use `/changelog full` to view the complete changelog."; + +/** Markdown generated from selected changelog entries and whether it hit a size cap. */ +export interface RenderedChangelog { + markdown: string; + truncated: boolean; +} + +/** Automatic startup changelog decision, including whether the marker should advance. */ +export interface StartupChangelogSelection { + markdown: string | undefined; + persistCurrentVersion: boolean; + truncated: boolean; + selectedEntries: number; +} + /** * Parse changelog entries from the file at `changelogPath`. Scans for `## [x.y.z]` * headings and collects each block until the next heading or EOF. @@ -87,19 +108,96 @@ export function compareVersions(v1: ChangelogEntry, v2: ChangelogEntry): number } /** - * Get entries newer than lastVersion + * Parse an omp changelog marker version into comparable parts. */ -export function getNewEntries(entries: ChangelogEntry[], lastVersion: string): ChangelogEntry[] { - // Parse lastVersion - const parts = lastVersion.split(".").map(Number); - const last: ChangelogEntry = { - major: parts[0] || 0, - minor: parts[1] || 0, - patch: parts[2] || 0, +export function parseChangelogVersion(version: string | undefined): ChangelogEntry | undefined { + const match = version?.match(/^(\d+)\.(\d+)\.(\d+)$/); + if (!match) { + return undefined; + } + + return { + major: Number.parseInt(match[1], 10), + minor: Number.parseInt(match[2], 10), + patch: Number.parseInt(match[3], 10), content: "", }; +} - return entries.filter(entry => compareVersions(entry, last) > 0); +/** + * Get entries newer than lastVersion. + */ +export function getNewEntries(entries: ChangelogEntry[], lastVersion: string): ChangelogEntry[] { + const parsedLastVersion = parseChangelogVersion(lastVersion); + if (!parsedLastVersion) { + return []; + } + + return entries.filter(entry => compareVersions(entry, parsedLastVersion) > 0); +} + +/** + * Render changelog entries newest-last and optionally cap the Markdown source size. + */ +export function renderChangelogEntries( + entries: ChangelogEntry[], + options: { maxBytes?: number; truncationHint?: string } = {}, +): RenderedChangelog { + const markdown = [...entries] + .reverse() + .map(entry => entry.content) + .join("\n\n"); + if (options.maxBytes === undefined || Buffer.byteLength(markdown) <= options.maxBytes) { + return { markdown, truncated: false }; + } + + const suffix = `\n\n…\n\n${options.truncationHint ?? STARTUP_CHANGELOG_FULL_HINT}`; + let low = 0; + let high = markdown.length; + while (low < high) { + const middle = Math.floor((low + high + 1) / 2); + if (Buffer.byteLength(markdown.slice(0, middle) + suffix) <= options.maxBytes) { + low = middle; + } else { + high = middle - 1; + } + } + + return { markdown: markdown.slice(0, low) + suffix, truncated: true }; +} + +/** + * Select bounded release notes for interactive startup. + */ +export function selectStartupChangelog( + entries: ChangelogEntry[], + lastVersion: string | undefined, + currentVersion: string, +): StartupChangelogSelection { + const parsedLastVersion = parseChangelogVersion(lastVersion); + if (!parsedLastVersion) { + return { markdown: undefined, persistCurrentVersion: true, truncated: false, selectedEntries: 0 }; + } + const markerVersion = lastVersion ?? ""; + if (markerVersion === currentVersion) { + return { markdown: undefined, persistCurrentVersion: false, truncated: false, selectedEntries: 0 }; + } + + const newEntries = getNewEntries(entries, markerVersion).slice(0, RECENT_CHANGELOG_ENTRY_LIMIT); + if (newEntries.length === 0) { + return { markdown: undefined, persistCurrentVersion: false, truncated: false, selectedEntries: 0 }; + } + + const rendered = renderChangelogEntries(newEntries, { + maxBytes: STARTUP_CHANGELOG_MAX_BYTES, + truncationHint: STARTUP_CHANGELOG_FULL_HINT, + }); + return { + markdown: rendered.markdown, + persistCurrentVersion: true, + truncated: rendered.truncated, + selectedEntries: newEntries.length, + }; } // Re-export getChangelogPath from paths.ts for convenience diff --git a/packages/coding-agent/test/utils/changelog.test.ts b/packages/coding-agent/test/utils/changelog.test.ts new file mode 100644 index 000000000..c8d43b3ff --- /dev/null +++ b/packages/coding-agent/test/utils/changelog.test.ts @@ -0,0 +1,232 @@ +/** + * Startup changelog contracts: + * + * - First-run/untrusted marker states persist the current version without + * replaying historical markdown. + * - Returning users only see a bounded startup slice (latest unseen releases, + * capped by source bytes), while explicit full changelog rendering remains + * unbounded. + * - The last-seen marker is a plain file in the agent dir. + */ + +import { describe, expect, test } from "bun:test"; +import { Buffer } from "node:buffer"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { removeWithRetries, VERSION } from "@oh-my-pi/pi-utils"; +import { + type ChangelogEntry, + RECENT_CHANGELOG_ENTRY_LIMIT, + readLastChangelogVersion, + renderChangelogEntries, + STARTUP_CHANGELOG_FULL_HINT, + STARTUP_CHANGELOG_MAX_BYTES, + selectStartupChangelog, + writeLastChangelogVersion, +} from "../../src/utils/changelog"; + +const CURRENT_VERSION = "2.0.0"; +const repoRoot = path.resolve(import.meta.dir, "..", "..", "..", ".."); +const cliEntry = path.join(repoRoot, "packages", "coding-agent", "src", "cli.ts"); +const packageDir = path.join(repoRoot, "packages", "coding-agent"); +const hasPtyHarness = + process.platform === "linux" && + (await Bun.file("/usr/bin/script").exists()) && + (await Bun.file("/usr/bin/timeout").exists()); +const PTY_STARTUP_OUTPUT_CEILING = 512 * 1024; + +function release(major: number, minor: number, patch: number, body: string): ChangelogEntry { + const heading = `## [${major}.${minor}.${patch}] - 2026-07-11`; + const content = `${heading}\n\n${body.trimEnd()}`; + return { major, minor, patch, content }; +} + +async function withTempAgentDir(callback: (agentDir: string) => Promise): Promise { + const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-changelog-marker-")); + try { + const result = await callback(agentDir); + return result; + } finally { + await removeWithRetries(agentDir); + } +} + +describe("selectStartupChangelog", () => { + const currentVersion = CURRENT_VERSION; + const history = [ + release(2, 0, 0, "### Added\n\n- Current release."), + release(1, 9, 0, "### Added\n\n- Previous release."), + release(1, 8, 0, "### Added\n\n- Older release."), + ]; + + test("treats missing, empty, malformed, and unreadable-equivalent markers as first run", () => { + const invalidMarkers: Array<{ name: string; value: string | undefined }> = [ + { name: "missing or unreadable marker", value: undefined }, + { name: "empty marker", value: "" }, + { name: "malformed marker", value: "not-a-semver" }, + { name: "incomplete marker", value: "1.9" }, + { name: "whitespace-padded marker", value: " 1.9.0 " }, + ]; + + for (const marker of invalidMarkers) { + const selection = selectStartupChangelog(history, marker.value, currentVersion); + expect(selection.markdown).toBeUndefined(); + expect(selection.persistCurrentVersion).toBe(true); + expect(selection.truncated).toBe(false); + expect(selection.selectedEntries).toBe(0); + } + }); + + test("does not render or rewrite when the marker already matches the current version", () => { + const selection = selectStartupChangelog(history, currentVersion, currentVersion); + + expect(selection.markdown).toBeUndefined(); + expect(selection.persistCurrentVersion).toBe(false); + expect(selection.truncated).toBe(false); + expect(selection.selectedEntries).toBe(0); + }); + + test("selects at most the three newest unseen releases for an older marker", () => { + const selection = selectStartupChangelog( + [ + release(1, 0, 5, "### Added\n\n- Unseen five."), + release(1, 0, 4, "### Added\n\n- Unseen four."), + release(1, 0, 3, "### Added\n\n- Unseen three."), + release(1, 0, 2, "### Added\n\n- Unseen two."), + release(1, 0, 1, "### Added\n\n- Unseen one."), + release(1, 0, 0, "### Added\n\n- Already seen."), + ], + "1.0.0", + "1.0.5", + ); + + expect(selection.persistCurrentVersion).toBe(true); + expect(selection.truncated).toBe(false); + expect(selection.selectedEntries).toBe(RECENT_CHANGELOG_ENTRY_LIMIT); + expect(selection.markdown).toContain("## [1.0.5]"); + expect(selection.markdown).toContain("## [1.0.4]"); + expect(selection.markdown).toContain("## [1.0.3]"); + expect(selection.markdown).not.toContain("## [1.0.2]"); + expect(selection.markdown).not.toContain("## [1.0.1]"); + expect(selection.markdown).not.toContain("## [1.0.0]"); + }); + + test("caps one oversized startup release and appends the full-changelog hint", () => { + const selection = selectStartupChangelog( + [release(2, 0, 0, `### Added\n\n- ${"x".repeat(STARTUP_CHANGELOG_MAX_BYTES * 2)}\nTAIL-ONE-RELEASE`)], + "1.0.0", + "2.0.0", + ); + + expect(selection.persistCurrentVersion).toBe(true); + expect(selection.selectedEntries).toBe(1); + expect(selection.truncated).toBe(true); + expect(selection.markdown).toContain(STARTUP_CHANGELOG_FULL_HINT); + expect(selection.markdown).not.toContain("TAIL-ONE-RELEASE"); + expect(Buffer.byteLength(selection.markdown ?? "")).toBeLessThanOrEqual(STARTUP_CHANGELOG_MAX_BYTES); + }); + + test("caps aggregate startup releases that exceed the byte budget and appends the full-changelog hint", () => { + const halfBudgetBody = "x".repeat(Math.ceil(STARTUP_CHANGELOG_MAX_BYTES / 2)); + const selection = selectStartupChangelog( + [ + release(1, 0, 4, `### Added\n\n- Four ${halfBudgetBody}\nTAIL-FOUR`), + release(1, 0, 3, `### Added\n\n- Three ${halfBudgetBody}\nTAIL-THREE`), + release(1, 0, 2, `### Added\n\n- Two ${halfBudgetBody}\nTAIL-TWO`), + release(1, 0, 1, "### Added\n\n- Already seen."), + ], + "1.0.1", + "1.0.4", + ); + + expect(selection.persistCurrentVersion).toBe(true); + expect(selection.selectedEntries).toBe(RECENT_CHANGELOG_ENTRY_LIMIT); + expect(selection.truncated).toBe(true); + expect(selection.markdown).toContain(STARTUP_CHANGELOG_FULL_HINT); + expect(selection.markdown).not.toContain("TAIL-FOUR"); + expect(Buffer.byteLength(selection.markdown ?? "")).toBeLessThanOrEqual(STARTUP_CHANGELOG_MAX_BYTES); + }); +}); + +describe("renderChangelogEntries", () => { + test("renders complete history when no maxBytes cap is passed", () => { + const largeBody = "y".repeat(STARTUP_CHANGELOG_MAX_BYTES); + const rendered = renderChangelogEntries([ + release(3, 0, 0, `### Added\n\n- Third ${largeBody}\nEND-THIRD`), + release(2, 0, 0, `### Added\n\n- Second ${largeBody}\nEND-SECOND`), + release(1, 0, 0, `### Added\n\n- First ${largeBody}\nEND-FIRST`), + ]); + + expect(rendered.truncated).toBe(false); + expect(rendered.markdown).toContain("END-FIRST"); + expect(rendered.markdown).toContain("END-SECOND"); + expect(rendered.markdown).toContain("END-THIRD"); + expect(rendered.markdown).not.toContain(STARTUP_CHANGELOG_FULL_HINT); + expect(Buffer.byteLength(rendered.markdown)).toBeGreaterThan(STARTUP_CHANGELOG_MAX_BYTES); + }); +}); + +describe("last changelog marker", () => { + test("reads a missing marker as undefined and writes the current version in the supplied agent dir", async () => { + await withTempAgentDir(async agentDir => { + expect(await readLastChangelogVersion(agentDir)).toBeUndefined(); + + await writeLastChangelogVersion(CURRENT_VERSION, agentDir); + + expect(await readLastChangelogVersion(agentDir)).toBe(CURRENT_VERSION); + expect(await Bun.file(path.join(agentDir, "last-changelog-version")).text()).toBe(CURRENT_VERSION); + }); + }); +}); + +describe.skipIf(!hasPtyHarness)("interactive startup changelog PTY smoke", () => { + test("does not dump packaged changelog history on first install with uncollapsed notes", async () => { + await withTempAgentDir(async agentDir => { + const root = await fs.mkdtemp(path.join(os.tmpdir(), "omp-changelog-pty-")); + try { + await fs.mkdir(path.join(root, "xdg-config"), { recursive: true }); + await fs.mkdir(path.join(root, "xdg-state"), { recursive: true }); + await fs.mkdir(path.join(root, "xdg-data"), { recursive: true }); + await Bun.write(path.join(agentDir, "config.yml"), "setupVersion: 1\ncollapseChangelog: false\n"); + + const proc = Bun.spawn( + ["timeout", "6s", "script", "-q", "-c", `bun ${JSON.stringify(cliEntry)}`, "/dev/null"], + { + cwd: repoRoot, + stdout: "pipe", + stderr: "pipe", + env: { + ...process.env, + HOME: root, + XDG_CONFIG_HOME: path.join(root, "xdg-config"), + XDG_STATE_HOME: path.join(root, "xdg-state"), + XDG_DATA_HOME: path.join(root, "xdg-data"), + PI_CODING_AGENT_DIR: agentDir, + PI_PACKAGE_DIR: packageDir, + PI_NO_TITLE: "1", + NO_COLOR: "1", + TERM: "xterm-256color", + }, + }, + ); + + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(proc.stdout).arrayBuffer(), + new Response(proc.stderr).text(), + proc.exited, + ]); + const output = Buffer.from(stdout).toString("utf8"); + + expect(exitCode).toBe(124); + expect(Buffer.byteLength(output)).toBeLessThan(PTY_STARTUP_OUTPUT_CEILING); + expect(output).not.toContain("## ["); + expect(output).not.toContain(STARTUP_CHANGELOG_FULL_HINT); + expect(stderr).not.toContain("Cannot find module"); + expect(await readLastChangelogVersion(agentDir)).toBe(VERSION); + } finally { + await removeWithRetries(root); + } + }); + }, 15_000); +}); From 449310eb1656397f49705b5d07509ebcab1a5da3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 11 Jul 2026 02:43:25 +0000 Subject: [PATCH 117/205] fix(coding-agent): kept startup changelog version current - Preserved newest-first ordering for startup changelog markdown so collapsed notices report the current release. - Kept default changelog rendering oldest-first for explicit recent/full views. - Added regression assertions for both startup and full-history heading order. --- packages/coding-agent/src/utils/changelog.ts | 11 +++++------ packages/coding-agent/test/utils/changelog.test.ts | 5 ++++- 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/utils/changelog.ts b/packages/coding-agent/src/utils/changelog.ts index fcc12be85..8fae797f4 100644 --- a/packages/coding-agent/src/utils/changelog.ts +++ b/packages/coding-agent/src/utils/changelog.ts @@ -137,16 +137,14 @@ export function getNewEntries(entries: ChangelogEntry[], lastVersion: string): C } /** - * Render changelog entries newest-last and optionally cap the Markdown source size. + * Render changelog entries oldest-first by default and optionally cap the Markdown source size. */ export function renderChangelogEntries( entries: ChangelogEntry[], - options: { maxBytes?: number; truncationHint?: string } = {}, + options: { maxBytes?: number; truncationHint?: string; oldestFirst?: boolean } = {}, ): RenderedChangelog { - const markdown = [...entries] - .reverse() - .map(entry => entry.content) - .join("\n\n"); + const orderedEntries = options.oldestFirst === false ? entries : [...entries].reverse(); + const markdown = orderedEntries.map(entry => entry.content).join("\n\n"); if (options.maxBytes === undefined || Buffer.byteLength(markdown) <= options.maxBytes) { return { markdown, truncated: false }; } @@ -191,6 +189,7 @@ export function selectStartupChangelog( const rendered = renderChangelogEntries(newEntries, { maxBytes: STARTUP_CHANGELOG_MAX_BYTES, truncationHint: STARTUP_CHANGELOG_FULL_HINT, + oldestFirst: false, }); return { markdown: rendered.markdown, diff --git a/packages/coding-agent/test/utils/changelog.test.ts b/packages/coding-agent/test/utils/changelog.test.ts index c8d43b3ff..e83b88b91 100644 --- a/packages/coding-agent/test/utils/changelog.test.ts +++ b/packages/coding-agent/test/utils/changelog.test.ts @@ -104,6 +104,7 @@ describe("selectStartupChangelog", () => { expect(selection.persistCurrentVersion).toBe(true); expect(selection.truncated).toBe(false); expect(selection.selectedEntries).toBe(RECENT_CHANGELOG_ENTRY_LIMIT); + expect(selection.markdown?.match(/## \[(\d+\.\d+\.\d+)\]/)?.[1]).toBe("1.0.5"); expect(selection.markdown).toContain("## [1.0.5]"); expect(selection.markdown).toContain("## [1.0.4]"); expect(selection.markdown).toContain("## [1.0.3]"); @@ -143,8 +144,9 @@ describe("selectStartupChangelog", () => { expect(selection.persistCurrentVersion).toBe(true); expect(selection.selectedEntries).toBe(RECENT_CHANGELOG_ENTRY_LIMIT); expect(selection.truncated).toBe(true); + expect(selection.markdown?.match(/## \[(\d+\.\d+\.\d+)\]/)?.[1]).toBe("1.0.4"); expect(selection.markdown).toContain(STARTUP_CHANGELOG_FULL_HINT); - expect(selection.markdown).not.toContain("TAIL-FOUR"); + expect(selection.markdown).not.toContain("TAIL-THREE"); expect(Buffer.byteLength(selection.markdown ?? "")).toBeLessThanOrEqual(STARTUP_CHANGELOG_MAX_BYTES); }); }); @@ -158,6 +160,7 @@ describe("renderChangelogEntries", () => { release(1, 0, 0, `### Added\n\n- First ${largeBody}\nEND-FIRST`), ]); + expect(rendered.markdown.match(/## \[(\d+\.\d+\.\d+)\]/)?.[1]).toBe("1.0.0"); expect(rendered.truncated).toBe(false); expect(rendered.markdown).toContain("END-FIRST"); expect(rendered.markdown).toContain("END-SECOND"); From 1c6f5dc18fd4f52b11b99670a25e29c0711f2c96 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 05:42:04 +0200 Subject: [PATCH 118/205] feat(coding-agent/prompts): refined agent delegation logic and constraints - Clarified that the agent must handle top-level scoping, planning, and cross-slice contracts before delegating work. - Restricted subagent usage to scenarios with genuine parallelism, discouraging "spawn-one-then-wait" patterns. - Updated criteria for inline work to include cases with only a single runnable slice, preventing unnecessary handoffs. --- packages/coding-agent/CHANGELOG.md | 6 ++++++ .../coding-agent/src/prompts/system/eager-task.md | 4 ++-- .../coding-agent/src/prompts/system/system-prompt.md | 11 +++++++++-- 3 files changed, 17 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d8041c95c..ea88cd5fc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,12 @@ ## [Unreleased] +### Changed + +- Refined agent delegation logic to prioritize top-level planning and scoping by the primary agent +- Optimized subagent usage to discourage single-agent delegation and improve parallel execution flows +- Clarified that prerequisite work for subagent tasks should be handled inline by the main agent + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/coding-agent/src/prompts/system/eager-task.md b/packages/coding-agent/src/prompts/system/eager-task.md index ab598dc26..91e8b49ae 100644 --- a/packages/coding-agent/src/prompts/system/eager-task.md +++ b/packages/coding-agent/src/prompts/system/eager-task.md @@ -1,7 +1,7 @@ Task delegation is enabled — subagents are the default for this request. -Explore and settle the approach FIRST. Once the design is settled, you MUST fan the work out to `{{toolRefs.task}}` subagents instead of implementing it yourself.{{#if taskBatch}} Batch independent slices into ONE parallel `{{toolRefs.task}}` call; never serialize work that can run concurrently.{{/if}} +Explore and settle the approach FIRST — scoping, top-level decomposition, and cross-slice contracts are YOUR job; NEVER spawn a subagent to produce the overall plan (per-slice design travels with its executor). Once the design is settled, you MUST fan the work out to `{{toolRefs.task}}` subagents instead of implementing it yourself.{{#if taskBatch}} Batch independent slices into ONE parallel `{{toolRefs.task}}` call; never serialize work that can run concurrently.{{/if}} -Work alone only for: a single-file edit under ~30 lines, a direct answer requiring no code changes, or a command the user explicitly asked you to run. +Work alone for: a single-file edit under ~30 lines, a direct answer requiring no code changes, a command the user explicitly asked you to run, or when only ONE runnable slice exists — a lone subagent is a lossy handoff, not parallelism. diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 78c1736f5..1e45733ed 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -163,12 +163,19 @@ Everything else—multi-file changes, refactors, new features, tests, investigat - Use `{{toolRefs.task}}` to map unknown code instead of reading file after file yourself. - NEVER abandon phases under scope pressure—delegate, don't shrink. - Default to parallel for complex changes. Delegate via `{{toolRefs.task}}` for non-importing file edits, multi-subsystem investigation, and decomposable work. -- **Maximize parallelism:** Break work into the widest possible {{#if taskBatch}}array of `tasks[]`{{else}}set of parallel `task` calls{{/if}}. NEVER serialize work that can run concurrently. Tasks touching different files or independent refactors should run in parallel; agents resolve their own file collisions live. +{{/if}} + +## Delegation gates: +- **Scope before you spawn.** YOU read the request, map the work, and name the independent slices. Delegation is NEVER the first move on a fresh request — unless the user already enumerated 2+ self-contained runnable slices, in which case dispatch them immediately in one batch. +- **NEVER outsource the top-level plan.** Scoping the request, the overall decomposition, and cross-slice contracts (formats, schemas, interfaces) are YOUR job. A generic "plan"/"design" subagent as step one starts blank, knows less than you, runs alone, and adds a full round-trip for ZERO parallelism — the canonical dumb spawn. Delegating design WITHIN a slice is fine: each executor details its own slice, and once the top-level split is settled you MAY fan out per-subsystem sub-planning in parallel. (Competing plans or independent reviews the user explicitly asked for are also legitimate.) +- **Spawn-one-then-wait is a bug.** A lone subagent you sit idle behind is you doing the work with extra latency plus a lossy handoff — do it inline. A single spawn is fine ONLY when you immediately continue another independent slice yourself, or it is a read-only scout keeping bulk exploration out of your context. +- **Width = real independence.** Fan out exactly as wide as the work genuinely decomposes{{#if taskBatch}}, batched into one `tasks[]` array{{else}}, as parallel calls in one message{{/if}}. NEVER serialize slices that can run concurrently; NEVER pad the batch with invented slices to look parallel. +- **Prerequisites run inline.** A step every slice depends on (shared schema, core interface, scaffold) has by definition nothing to run beside it — do it yourself, then fan out. "Parallelize" means parallel EXECUTION of the independent slices, not routing sequential steps through agents. +- **You own the user's intent.** Subagents never see this conversation. Interpreting the request and taste calls stay with you; each assignment carries every requirement its slice needs. {{#when MAX_CONCURRENCY ">" 0}} - **Concurrency cap:** At most {{pluralize MAX_CONCURRENCY "subagent" "subagents"}} run at once in this session — anything beyond that just queues, so a {{#if taskBatch}}`tasks[]` batch{{else}}set of parallel `task` calls{{/if}} larger than {{MAX_CONCURRENCY}} only delays results. Keep the fan-out at or under the cap. {{/when}} - **Sequence only when necessary:** The only reason to run A before B is if B strictly requires A's output to function (e.g., a core API contract or schema migration). {{#if taskIrcEnabled}}If the missing piece is small, run them in parallel and have B ask A via `irc`!{{/if}} -{{/if}} {{/has}} EXECUTION WORKFLOW From 2f97b7fe4b04f10dc34fc7451ecfa3355a70a4f7 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 05:36:37 +0200 Subject: [PATCH 119/205] feat(coding-agent): removed plan subagent - Removed the `plan` agent definition and associated prompt file from bundled agents. - Updated documentation to reflect the removal of `plan` from available subagents. - Cleaned up related tests to remove references to the deprecated agent. --- docs/task-agent-discovery.md | 4 +- docs/tools/task.md | 2 +- packages/coding-agent/CHANGELOG.md | 5 ++ .../coding-agent/src/prompts/agents/plan.md | 47 ------------------- packages/coding-agent/src/task/agents.ts | 2 - .../test/bundled-agent-parsing.test.ts | 33 +++++-------- .../tools/task-agent-capabilities.test.ts | 4 +- 7 files changed, 21 insertions(+), 76 deletions(-) delete mode 100644 packages/coding-agent/src/prompts/agents/plan.md diff --git a/docs/task-agent-discovery.md b/docs/task-agent-discovery.md index 7e35066e6..c725f61f0 100644 --- a/docs/task-agent-discovery.md +++ b/docs/task-agent-discovery.md @@ -35,7 +35,7 @@ Parsing comes from frontmatter via `parseAgentFields()` (`src/discovery/helpers. - `spawns` accepts `*`, CSV, or array - backward-compat behavior: if `spawns` missing but `tools` includes `task`, `spawns` becomes `*` - `output` is passed through as opaque schema data -- `read-summarize: false` (parsed as `readSummarize`) forces the subagent's `read` tool to return verbatim file content instead of structural summaries — `runSubprocess` applies it as a `read.summarize.enabled: false` override on the subagent's isolated settings (`src/task/executor.ts`). `explore` and `librarian` ship with it disabled. Defaults to enabled when the field is absent. +- `read-summarize: false` (parsed as `readSummarize`) forces the subagent's `read` tool to return verbatim file content instead of structural summaries — `runSubprocess` applies it as a `read.summarize.enabled: false` override on the subagent's isolated settings (`src/task/executor.ts`). `scout` and `librarian` ship with it disabled. Defaults to enabled when the field is absent. ## Bundled agents @@ -43,7 +43,7 @@ Bundled agents are embedded at build time (`src/task/agents.ts`) using text impo `EMBEDDED_AGENT_DEFS` defines: -- `explore`, `plan`, `designer`, `reviewer`, `librarian`, `oracle` from prompt files +- `scout`, `designer`, `reviewer`, `librarian` from prompt files - `task` and `sonic` from shared `task.md` body plus injected frontmatter Loading path: diff --git a/docs/tools/task.md b/docs/tools/task.md index abf8bdcbe..790fc14ce 100644 --- a/docs/tools/task.md +++ b/docs/tools/task.md @@ -106,7 +106,7 @@ Artifacts and side channels: - off — single spawn per call; `tasks`/`context` are rejected and removed from the schema. - Isolation mode (`task.isolation.mode`): `none`, `auto`, `apfs`, `btrfs`, `zfs`, `reflink`, `overlayfs`, `projfs`, `block-clone`, `rcopy` (legacy `worktree`, `fuse-overlay`, `fuse-projfs` accepted for back-compat); the PAL resolves the actual backend with fallback. - Isolation merge strategy: patch mode (capture/apply root patches) or branch mode (commit to `omp/task/`, cherry-pick into parent). -- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `sonic`, `librarian`, `oracle`). +- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`scout`, `designer`, `reviewer`, `task`, `sonic`, `librarian`). ## Side Effects - Filesystem diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ea88cd5fc..4223ac799 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -8,6 +8,11 @@ - Optimized subagent usage to discourage single-agent delegation and improve parallel execution flows - Clarified that prerequisite work for subagent tasks should be handled inline by the main agent +### Removed + +- Removed the bundled `plan` subagent from available task agents +- Removed the bundled `plan` subagent from available task agents. + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/coding-agent/src/prompts/agents/plan.md b/packages/coding-agent/src/prompts/agents/plan.md deleted file mode 100644 index 66ca395b0..000000000 --- a/packages/coding-agent/src/prompts/agents/plan.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -name: plan -description: Software architect for complex multi-file architectural decisions. NOT for simple tasks, single-file changes, or tasks completable in <5 tool calls. -tools: read, grep, glob, bash, lsp, web_search, ast_grep -spawns: scout -model: pi/plan, pi/slow ---- - -Analyze the codebase and the user's request. Produce a detailed implementation plan. - -## Phase 1: Understand -1. Parse requirements precisely -2. Identify ambiguities; list assumptions - -## Phase 2: Explore -1. Find existing patterns via `grep`/`glob` -2. Read key files; understand architecture -3. Trace data flow through relevant paths -4. Identify types, interfaces, contracts -5. Note dependencies between components - -You MUST spawn `scout` agents for independent areas and synthesize findings. - -## Phase 3: Design -1. List concrete changes (files, functions, types) -2. Define sequence and dependencies -3. Identify edge cases and error conditions -4. Consider alternatives; justify your choice -5. Note pitfalls/tricky parts - -## Phase 4: Produce Plan - -You MUST write a plan executable without re-exploration. - - -- **Summary**: What to build and why (one paragraph). -- **Changes**: Concrete changes (files, functions, types). Exact file paths/line ranges where relevant. -- **Sequence**: Ordering and dependencies between sub-tasks. -- **Edge Cases**: Edge cases and error conditions to watch. -- **Verification**: Steps to verify correctness. -- **Critical Files**: Files the implementer must read to understand the codebase. - - - -You MUST operate as read-only. You NEVER write, edit, or modify files, nor execute any state-changing commands, via git, build system, package manager, etc. -You MUST keep going until complete. - diff --git a/packages/coding-agent/src/task/agents.ts b/packages/coding-agent/src/task/agents.ts index 4399f2751..627a99304 100644 --- a/packages/coding-agent/src/task/agents.ts +++ b/packages/coding-agent/src/task/agents.ts @@ -10,7 +10,6 @@ import designerMd from "../prompts/agents/designer.md" with { type: "text" }; // Embed agent markdown files at build time import agentFrontmatterTemplate from "../prompts/agents/frontmatter.md" with { type: "text" }; import librarianMd from "../prompts/agents/librarian.md" with { type: "text" }; -import planMd from "../prompts/agents/plan.md" with { type: "text" }; import reviewerMd from "../prompts/agents/reviewer.md" with { type: "text" }; import scoutMd from "../prompts/agents/scout.md" with { type: "text" }; import taskMd from "../prompts/agents/task.md" with { type: "text" }; @@ -41,7 +40,6 @@ function buildAgentContent(def: EmbeddedAgentDef): string { const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ { fileName: "scout.md", template: scoutMd }, - { fileName: "plan.md", template: planMd }, { fileName: "designer.md", template: designerMd }, { fileName: "reviewer.md", template: reviewerMd }, { fileName: "librarian.md", template: librarianMd }, diff --git a/packages/coding-agent/test/bundled-agent-parsing.test.ts b/packages/coding-agent/test/bundled-agent-parsing.test.ts index a4da8e57d..87133e0fd 100644 --- a/packages/coding-agent/test/bundled-agent-parsing.test.ts +++ b/packages/coding-agent/test/bundled-agent-parsing.test.ts @@ -15,22 +15,13 @@ describe("bundled agent parsing", () => { expect(reviewer?.thinkingLevel).toBeUndefined(); }); - it("lets plan inherit thinking effort from its model role", () => { - const plan = getBundledAgent("plan"); - - expect(plan).toBeDefined(); - expect(plan?.source).toBe("bundled"); - expect(plan?.model).toEqual(["pi/plan", "pi/slow"]); - expect(plan?.thinkingLevel).toBeUndefined(); - }); - // Issue #4761: with `modelRoles.slow: ...:xhigh`, the role's explicit effort // suffix must survive agent-pattern expansion and model resolution for the // bundled agents routed at that role. The executor picks // `agent.thinkingLevel ?? resolvedThinkingLevel` (task/executor.ts), so a - // bundled frontmatter pin would mask the suffix — reviewer/plan declare none + // bundled frontmatter pin would mask the suffix — reviewer declares none // (asserted above) and the resolved level below is what the subagent runs at. - it("resolves the configured slow-role effort suffix for reviewer and plan", () => { + it("resolves the configured slow-role effort suffix for reviewer", () => { const gpt55 = buildModel({ id: "gpt-5.5", name: "GPT-5.5 Codex", @@ -45,19 +36,17 @@ describe("bundled agent parsing", () => { maxTokens: 128000, }); const settings = Settings.isolated({ - modelRoles: { slow: "openai-codex/gpt-5.5:xhigh", plan: "openai-codex/gpt-5.5:xhigh" }, + modelRoles: { slow: "openai-codex/gpt-5.5:xhigh" }, }); const registry = { getAvailable: () => [gpt55] } as Parameters[1]; - for (const name of ["reviewer", "plan"]) { - const agent = getBundledAgent(name); - expect(agent?.thinkingLevel).toBeUndefined(); - const patterns = resolveAgentModelPatterns({ agentModel: agent?.model, settings }); - const resolved = resolveModelOverride(patterns, registry, settings); - expect(resolved.model?.provider).toBe("openai-codex"); - expect(resolved.model?.id).toBe("gpt-5.5"); - expect(resolved.thinkingLevel).toBe(Effort.XHigh); - expect(resolved.explicitThinkingLevel).toBe(true); - } + const agent = getBundledAgent("reviewer"); + expect(agent?.thinkingLevel).toBeUndefined(); + const patterns = resolveAgentModelPatterns({ agentModel: agent?.model, settings }); + const resolved = resolveModelOverride(patterns, registry, settings); + expect(resolved.model?.provider).toBe("openai-codex"); + expect(resolved.model?.id).toBe("gpt-5.5"); + expect(resolved.thinkingLevel).toBe(Effort.XHigh); + expect(resolved.explicitThinkingLevel).toBe(true); }); }); diff --git a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts index 62011854a..022b2f057 100644 --- a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts +++ b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts @@ -14,7 +14,7 @@ describe("task agent capability descriptions", () => { const agents = loadBundledAgents(); expect(isReadOnlyAgent(agentByName(agents, "scout"))).toBe(true); - for (const name of ["task", "sonic", "plan", "reviewer", "designer"]) { + for (const name of ["task", "sonic", "reviewer", "designer"]) { expect(isReadOnlyAgent(agentByName(agents, name))).toBe(false); } }); @@ -24,7 +24,7 @@ describe("task agent capability descriptions", () => { expect(agentByName(agents, "scout").readSummarize).toBe(false); expect(agentByName(agents, "librarian").readSummarize).toBe(false); - for (const name of ["task", "sonic", "plan", "reviewer", "designer"]) { + for (const name of ["task", "sonic", "reviewer", "designer"]) { expect(agentByName(agents, name).readSummarize).toBeUndefined(); } }); From 45143e8c751739b41e9770512a91b743640ded50 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 03:17:05 +0200 Subject: [PATCH 120/205] feat(natives): restricted traversal depth for glob pattern matching - Added `walk_depth_bound` to calculate the maximum file path depth based on glob pattern segments. - Configured the glob walker to restrict traversal depth for non-recursive patterns, preventing unnecessary scanning of subtrees that cannot match. - Improved response times for narrow globs over large directories by pruning the search space earlier. - Added test cases to verify depth-bounded matching behavior and pattern edge cases. --- crates/pi-natives/src/glob.rs | 52 +++++++++++++++++++++++++++++- crates/pi-natives/src/glob_util.rs | 39 ++++++++++++++++++++++ packages/natives/CHANGELOG.md | 4 +++ 3 files changed, 94 insertions(+), 1 deletion(-) diff --git a/crates/pi-natives/src/glob.rs b/crates/pi-natives/src/glob.rs index e30d49ec4..92e7f81a4 100644 --- a/crates/pi-natives/src/glob.rs +++ b/crates/pi-natives/src/glob.rs @@ -172,6 +172,9 @@ fn run_glob( ct: task::CancelToken, ) -> Result { let walk_glob_pattern = glob_util::build_glob_pattern(&config.pattern, config.recursive); + // Non-recursive patterns bound the walk: `dir/*` must not traverse the + // entire subtree under `dir` to match only direct children. + let walk_depth_limit = glob_util::walk_depth_bound(&walk_glob_pattern); let walk_glob = pi_walker::CompiledWalkGlob::new([walk_glob_pattern]) .map_err(|err| Error::from_reason(format!("Invalid glob pattern: {err}")))?; if config.max_results == 0 { @@ -192,7 +195,7 @@ fn run_glob( .detail(scan_detail) .order(pi_walker::WalkOrder::Path) .emit_root(false) - .depth(1, usize::MAX) + .depth(1, walk_depth_limit) .directory_errors(pi_walker::DirectoryErrorMode::SkipSkippable) .cache(config.cache) .empty_recheck(pi_walker::EmptyRecheck::Configured) @@ -378,4 +381,51 @@ mod tests { "gitignored directory should be pruned before matching, got {paths:?}" ); } + + #[test] + fn run_glob_depth_bounded_patterns_still_match_at_their_exact_depth() { + // The walk for non-`**` patterns is depth-bounded (see walk_depth_bound); + // this defends the boundary: matches AT the bound depth must survive, + // deeper entries must not appear, and the mtime-ranked mode (the glob + // tool default) must behave identically to the streaming mode. + let root = TempDirGuard::new(); + fs::write(root.path().join("top.txt"), "top").expect("write top file"); + fs::create_dir_all(root.path().join("deep/nested")).expect("create nested dirs"); + fs::write(root.path().join("deep/child.txt"), "mid").expect("write mid file"); + fs::write(root.path().join("deep/nested/leaf.txt"), "leaf").expect("write leaf file"); + + let run = |pattern: &str| { + super::run_glob( + super::GlobConfig { + root: root.path().to_path_buf(), + pattern: pattern.to_string(), + recursive: false, + include_hidden: true, + file_type_filter: None, + max_results: 100, + use_gitignore: true, + mentions_node_modules: false, + sort_by_mtime: true, + cache: false, + }, + None, + crate::task::CancelToken::default(), + ) + .expect("glob succeeds") + }; + + let direct = run("*.txt"); + assert_eq!(match_paths(&direct), ["top.txt"]); + + let two_deep = run("deep/*.txt"); + assert_eq!(match_paths(&two_deep), ["deep/child.txt"]); + + let wildcard_dir = run("*/nested/leaf.txt"); + assert_eq!(match_paths(&wildcard_dir), ["deep/nested/leaf.txt"]); + + let recursive = run("**/*.txt"); + let mut recursive_paths = match_paths(&recursive); + recursive_paths.sort_unstable(); + assert_eq!(recursive_paths, ["deep/child.txt", "deep/nested/leaf.txt", "top.txt"]); + } } diff --git a/crates/pi-natives/src/glob_util.rs b/crates/pi-natives/src/glob_util.rs index 9ee36bb99..489376e50 100644 --- a/crates/pi-natives/src/glob_util.rs +++ b/crates/pi-natives/src/glob_util.rs @@ -64,6 +64,29 @@ pub fn build_glob_pattern(glob: &str, recursive: bool) -> String { fix_unclosed_braces(pattern) } +/// Maximum walk depth (path components) a normalized glob pattern can match, +/// or `usize::MAX` when unbounded. +/// +/// Walk-relative globs compile with `literal_separator(true)`, so `*`, `?`, +/// and `[...]` never cross `/` — a pattern with N literal segments can only +/// match entries at most N components deep. Bounding the walk to that depth +/// keeps non-recursive patterns (`*`, `dir/*.json`) from traversing an entire +/// subtree they can never match into (the source of "narrow glob timed out on +/// a populated directory" failures). +/// +/// `**` matches any number of components and `{...}` alternations may contain +/// `/`, so both disable the bound. +pub fn walk_depth_bound(pattern: &str) -> usize { + if pattern.contains("**") || pattern.contains('{') { + return usize::MAX; + } + pattern + .split('/') + .filter(|seg| !seg.is_empty()) + .count() + .max(1) +} + /// Compile a glob pattern string into a [`CompiledGlob`]. /// /// When `recursive` is true, simple patterns (no path separators, no leading @@ -220,6 +243,22 @@ mod tests { assert_eq!(build_glob_pattern("*.ts", false), "*.ts"); } + #[test] + fn walk_depth_bound_counts_segments_for_bounded_patterns() { + assert_eq!(walk_depth_bound("*"), 1); + assert_eq!(walk_depth_bound("*.json"), 1); + assert_eq!(walk_depth_bound("dir/*.ts"), 2); + assert_eq!(walk_depth_bound("a/*/c.txt"), 3); + } + + #[test] + fn walk_depth_bound_unbounded_for_recursive_and_brace_patterns() { + assert_eq!(walk_depth_bound("**/*"), usize::MAX); + assert_eq!(walk_depth_bound("src/**/*.ts"), usize::MAX); + // `{}` groups may contain `/`, so segment counting is unsound for them. + assert_eq!(walk_depth_bound("{a/b,c}/d.txt"), usize::MAX); + } + #[test] fn compiled_non_recursive_extension_glob_matches_only_root_files() { let glob = compile_glob("*.rs", false).expect("compile non-recursive extension glob"); diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index fb1e08bea..cdf7cff61 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed non-recursive glob patterns traversing entire subtrees they could never match into: `dir/*.json` walked everything under `dir` (unbounded depth) because only the match filter — not the walk — knew the pattern was shallow. The walker is now depth-bounded by the pattern's segment count when it contains no `**` or brace alternation (wildcards never cross `/`), so narrow direct-child globs over huge directories (`~/.cache/*`-style) return in milliseconds instead of hitting the 5s timeout with zero partial matches. + ## [16.3.13] - 2026-07-09 ### Fixed From 83fbefac29f6a81fe86656ea7212bfd41b2c1568 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 03:17:06 +0200 Subject: [PATCH 121/205] fix(coding-agent): disabled context padding for raw file reads - Disable context expansion in raw mode to ensure verbatim content extraction. - Enforce strict adherence to requested line ranges when raw selectors are used. - Remove padding in range calculations to prevent indistinguishable context lines in raw output. --- packages/coding-agent/src/tools/read.ts | 26 ++++++++++++++++--------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 511e59086..e26f13457 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -1289,13 +1289,17 @@ export class ReadTool implements AgentTool { const ignoreResultLimits = options.ignoreResultLimits ?? false; const requestedEnd = limit !== undefined ? Math.min(requestedStart + limit, allLines.length) : allLines.length; // Expand only on sides the user actually constrained: leading context - // when offset>1, trailing context when a finite limit was set. + // when offset>1, trailing context when a finite limit was set. Raw mode + // never expands — without line numbers the padding is indistinguishable + // from requested content, so `raw:31-31` must return line 31 and nothing + // else (verbatim-extraction contract). + const rawDisplay = options.raw === true; const expanded = expandRangeWithContext( requestedStart, requestedEnd, allLines.length, - offset !== undefined && offset > 1, - limit !== undefined, + !rawDisplay && offset !== undefined && offset > 1, + !rawDisplay && limit !== undefined, ); const startLine = expanded.startLine; const endLineExpanded = expanded.endLine; @@ -2527,10 +2531,14 @@ export class ReadTool implements AgentTool { } // User-requested 0-indexed range start. Lines BEFORE this become - // leading context (added below if offset is explicit). + // leading context (added below if offset is explicit). Raw mode + // never adds context: without line numbers the padding is + // indistinguishable from requested content, so `raw:31-31` must + // return line 31 and nothing else. + const rawSelector = isRawSelector(parsed); const requestedStart = offset ? Math.max(0, offset - 1) : 0; - const expandStart = offset !== undefined && offset > 1; - const expandEnd = limit !== undefined; + const expandStart = !rawSelector && offset !== undefined && offset > 1; + const expandEnd = !rawSelector && limit !== undefined; const leadingContext = expandStart ? Math.min(requestedStart, RANGE_LEADING_CONTEXT_LINES) : 0; const trailingContext = expandEnd ? RANGE_TRAILING_CONTEXT_LINES : 0; const startLine = requestedStart - leadingContext; @@ -2581,7 +2589,6 @@ export class ReadTool implements AgentTool { // verbatim bytes for paste-back-into-tool workflows. Total byte/line // counts in `truncation` keep reflecting the source, not the trimmed // view — column truncation surfaces separately via `.limits()`. - const rawSelector = isRawSelector(parsed); const maxColumns = resolveOutputMaxColumns(this.session.settings); // Column truncation is display-only. `collectedLines` MUST stay // byte-for-byte with the on-disk content so the snapshot recorded @@ -2972,8 +2979,9 @@ export class ReadTool implements AgentTool { const { offset, limit } = selToOffsetLimit(parsedSel); const requestedStart = offset ? Math.max(0, offset - 1) : 0; - const expandStart = offset !== undefined && offset > 1; - const expandEnd = limit !== undefined; + // Raw mode never adds context lines — see the plain-file range path. + const expandStart = !rawSelector && offset !== undefined && offset > 1; + const expandEnd = !rawSelector && limit !== undefined; const leadingContext = expandStart ? Math.min(requestedStart, RANGE_LEADING_CONTEXT_LINES) : 0; const trailingContext = expandEnd ? RANGE_TRAILING_CONTEXT_LINES : 0; const startLine = requestedStart - leadingContext; From d993b13c80d96271fc6e32dafd8d517f87f76cf6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 03:17:06 +0200 Subject: [PATCH 122/205] fix(coding-agent): strengthened browser interaction reliability and error transparency - Implemented a try/catch envelope for browser evaluations to surface detailed page-side JS exceptions and detect unsupported Promise returns. - Added transparent error reporting for cmux browser surface limitations regarding screenshot clipping and full-page captures. - Forced tab activation before screenshot capture to prevent stale or sibling-tab image data in shared-endpoint environments. --- .../src/tools/browser/cmux/cmux-tab.ts | 39 ++++++++++++--- .../src/tools/browser/cmux/rpc.ts | 50 +++++++++++++++++++ .../src/tools/browser/tab-worker.ts | 7 +++ 3 files changed, 89 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts b/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts index a0e311d29..eeb49af88 100644 --- a/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts +++ b/packages/coding-agent/src/tools/browser/cmux/cmux-tab.ts @@ -24,7 +24,8 @@ import { cmuxSnapshotToObservation, GEOMETRY_SCRIPT, mapWaitUntil, - serializeEval, + serializeEvalWithEnvelope, + unwrapEvalEnvelope, } from "./rpc"; import type { CmuxSocketClient } from "./socket-client"; @@ -445,10 +446,14 @@ export class CmuxTab { fn: string | ((...args: TArgs) => TResult | Promise), ...args: TArgs ): Promise { - const result = (await this.#request("browser.eval", { - script: serializeEval(fn as string | ((...args: unknown[]) => unknown), args), - })) as CmuxEvalResult; - return result.value as TResult; + // A script that throws inside the daemon comes back as a bare + // `js_error: A JavaScript exception occurred` with no message or stack. + // Catch page-side instead so the exception is diagnosable, and turn the + // daemon's other blind spot — Promise return values it cannot + // serialize — into an actionable error instead of "unsupported type". + const script = serializeEvalWithEnvelope(fn as string | ((...args: unknown[]) => unknown), args); + const result = (await this.#request("browser.eval", { script })) as CmuxEvalResult; + return unwrapEvalEnvelope(result.value, "tab.evaluate()"); } async scrollIntoView(selector: string): Promise { @@ -479,10 +484,22 @@ export class CmuxTab { async screenshot(opts: ScreenshotOptions = {}): Promise { const context = this.#requireRunContext("tab.screenshot()"); + // The cmux daemon's `browser.screenshot` captures the surface viewport + // only — it has no element-clip or full-page mode, and Bun.Image cannot + // crop locally. Degrade transparently instead of silently mislabeling + // the capture: scroll the element into view, then TELL the model the + // image is the full viewport (reports showed selector captures being + // consumed as element crops). + const captureNotes: string[] = []; if (opts.selector) { await this.scrollIntoView(opts.selector); + captureNotes.push( + `selector ${JSON.stringify(opts.selector)} was scrolled into view, but this surface cannot clip to an element — the image is the full viewport`, + ); + } + if (opts.fullPage) { + captureNotes.push("fullPage is unavailable on this surface — the image is the viewport only"); } - void opts.fullPage; const result = await this.#captureScreenshotPng(context.timeoutMs); const buffer = Buffer.from(result.png_base64, "base64"); const captureMime = "image/png"; @@ -528,6 +545,9 @@ export class CmuxTab { dest, resized, }); + if (captureNotes.length > 0) { + lines.push(`[cmux surface: ${captureNotes.join("; ")}]`); + } context.displays.push({ type: "text", text: lines.join("\n") }); context.displays.push({ type: "image", data: resized.data, mimeType: resized.mimeType }); } @@ -724,7 +744,12 @@ export class CmuxTab { const callable = (0, eval)("(" + source + ")"); return callable(element, ...args); })()`; - return await this.#evalScript(script); + // Envelope so a stale selector or a throwing callback reports its actual + // error instead of the daemon's generic js_error (see tab.evaluate()). + const result = (await this.#request("browser.eval", { + script: serializeEvalWithEnvelope(script, []), + })) as CmuxEvalResult; + return unwrapEvalEnvelope(result.value, "elementHandle.evaluate()"); } async pageContent(): Promise { diff --git a/packages/coding-agent/src/tools/browser/cmux/rpc.ts b/packages/coding-agent/src/tools/browser/cmux/rpc.ts index ed965ea78..b7722a90b 100644 --- a/packages/coding-agent/src/tools/browser/cmux/rpc.ts +++ b/packages/coding-agent/src/tools/browser/cmux/rpc.ts @@ -1,3 +1,4 @@ +import { ToolError } from "../../tool-errors"; import type { Observation, ObservationEntry } from "../tab-protocol"; export interface CmuxKind { @@ -120,6 +121,55 @@ export function serializeEval(fn: string | ((...args: unknown[]) => unknown), ar return `(${fn.toString()})(${args.map(arg => JSON.stringify(arg)).join(",")})`; } +/** + * Like {@link serializeEval}, but wraps the expression in a page-side + * try/catch envelope so a throwing script surfaces its message + stack + * instead of the daemon's opaque `js_error: A JavaScript exception occurred`, + * and a Promise return (which the daemon cannot serialize) is flagged + * explicitly rather than failing as "unsupported type". + * + * String scripts run through indirect eval to keep global-scope semantics; + * function sources are already expressions and are invoked directly. + * `undefined` results come back as `null` (JSON cannot carry `undefined`). + * Decode with {@link unwrapEvalEnvelope}. + */ +export function serializeEvalWithEnvelope(fn: string | ((...args: unknown[]) => unknown), args: unknown[]): string { + const inner = serializeEval(fn, args); + const expr = typeof fn === "string" ? `(0, eval)(${JSON.stringify(inner)})` : inner; + return `(() => { + try { + const __v = (${expr}); + if (__v && typeof __v.then === "function") return { __ompPromise: true }; + return { __ompOk: __v === undefined ? null : __v }; + } catch (e) { + return { __ompErr: (e && (e.stack || e.message)) || String(e) }; + } + })()`; +} + +/** + * Decode a {@link serializeEvalWithEnvelope} result: rethrow page-side + * exceptions as rich {@link ToolError}s, reject unserializable Promise + * returns with an actionable message, and pass through values from daemons + * that did not run the wrapper. + */ +export function unwrapEvalEnvelope(value: unknown, label: string): TResult { + if (value && typeof value === "object") { + if ("__ompErr" in value && typeof value.__ompErr === "string") { + throw new ToolError(`${label} threw a JavaScript exception:\n${value.__ompErr}`); + } + if ("__ompPromise" in value && value.__ompPromise === true) { + throw new ToolError( + `${label} returned a Promise, but this surface evaluates synchronously and cannot await it — return a plain value (poll with waitForFunction for async state instead)`, + ); + } + if ("__ompOk" in value) { + return value.__ompOk as TResult; + } + } + return value as TResult; +} + export function mapWaitUntil(waitUntil: string | undefined): "interactive" | "complete" { return waitUntil === "domcontentloaded" ? "interactive" : "complete"; } diff --git a/packages/coding-agent/src/tools/browser/tab-worker.ts b/packages/coding-agent/src/tools/browser/tab-worker.ts index 677db7a6f..6bc0217a3 100644 --- a/packages/coding-agent/src/tools/browser/tab-worker.ts +++ b/packages/coding-agent/src/tools/browser/tab-worker.ts @@ -1140,6 +1140,13 @@ export class WorkerCore { opts: ScreenshotOptions = {}, ): Promise { const page = this.#requirePage(); + // Multiple tabs can share one Chromium (sibling headless tabs on a shared + // endpoint, cdp/app attach). CDP `Page.captureScreenshot` reads the + // compositor surface, which follows the *active* target — a backgrounded + // page can stall waiting for a fresh frame (the 20s screenshot timeouts) + // or hand back a sibling tab's pixels. Activate first; best-effort so an + // already-active or freshly-closed target never fails the capture. + await untilAborted(signal, () => page.bringToFront()).catch(() => undefined); const fullPage = opts.selector ? false : (opts.fullPage ?? false); // An explicit save path picks the full-res capture format: puppeteer encodes // png/jpeg/webp natively, so `save: "shot.webp"` gets real WebP bytes instead From 530faffd23e327cc5ead30df7cba0024550fd458 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 03:17:06 +0200 Subject: [PATCH 123/205] fix(coding-agent): clarified timeout status in glob scans to prevent misleading empty results - Removed definitive "No files found" message when glob scans time out to prevent misleading claims of file absence. - Updated timeout notice to explicitly state that results are incomplete and provide actionable guidance on scoping the search. - Modified TUI renderer to display "No matches before timeout (scan incomplete)" status instead of "No files found" for timed-out scans. --- packages/coding-agent/CHANGELOG.md | 8 ++++++++ packages/coding-agent/src/tools/glob.ts | 25 +++++++++++++++++++------ 2 files changed, 27 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4223ac799..52ff045e4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,14 @@ - Removed the bundled `plan` subagent from available task agents - Removed the bundled `plan` subagent from available task agents. +### Fixed + +- Fixed `glob` reporting the contradictory "No files found matching pattern" next to a "timed out; returning 0 partial matches" notice. A timed-out empty scan now states explicitly that the result is incomplete (not proof of absence) and suggests scoping to a deeper directory, and the TUI renders it as "No matches before timeout (scan incomplete)" instead of a definitive no-files claim. +- Fixed `read` adding invisible context padding to raw range selectors: `raw:31-31` returned lines 30–34 (1 leading + 3 trailing context lines) with nothing to distinguish the padding, corrupting verbatim-extraction workflows. Raw ranges now return exactly the requested lines across all three range paths (plain files, in-memory/bridge reads, artifacts); numbered reads keep the self-describing padding. +- Fixed browser screenshots capturing the wrong tab or hanging until the op timeout when multiple tabs share one Chromium (sibling headless tabs, `cdp_url`/app attach): CDP reads the *active* target's compositor surface, so a backgrounded page could stall waiting for a frame or return a sibling's pixels. The worker now activates the page (`bringToFront`) before every capture, best-effort. +- Fixed cmux `tab.screenshot({ selector })` silently returning a full-viewport capture that models consumed as an element crop. The cmux daemon has no element-clip or full-page capture; the tool still scrolls the selector into view but now labels the image as full-viewport (same for `fullPage`) instead of mislabeling it. +- Fixed cmux `tab.evaluate()` / `elementHandle.evaluate()` errors surfacing as the daemon's opaque `js_error: A JavaScript exception occurred`. Scripts now run inside a page-side try/catch envelope that returns the real message and stack, and a Promise return (which the daemon cannot serialize) yields an actionable "evaluates synchronously" error instead of an unsupported-type failure. + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/coding-agent/src/tools/glob.ts b/packages/coding-agent/src/tools/glob.ts index 79176c459..00f0a2329 100644 --- a/packages/coding-agent/src/tools/glob.ts +++ b/packages/coding-agent/src/tools/glob.ts @@ -258,7 +258,7 @@ export class GlobTool implements AgentTool { const buildResult = ( files: string[], - opts?: { notice?: string; forceTruncated?: boolean }, + opts?: { notice?: string; forceTruncated?: boolean; timedOut?: boolean }, ): AgentToolResult => { const notice = opts?.notice; const forceTruncated = opts?.forceTruncated ?? false; @@ -271,7 +271,10 @@ export class GlobTool implements AgentTool { cwd: this.session.cwd, missingPaths: missingPaths.length > 0 ? missingPaths : undefined, }; - const parts = ["No files found matching pattern"]; + // A timed-out empty result is an incomplete scan, not a verified + // absence — never emit the definitive "No files found" claim next + // to a timeout notice (the two statements contradict each other). + const parts = opts?.timedOut ? [] : ["No files found matching pattern"]; if (notice) parts.push(notice); if (missingPathsNote) parts.push(missingPathsNote); // Zero results is useless regardless of notices: the follow-up @@ -452,8 +455,15 @@ export class GlobTool implements AgentTool { partial.sort((a, b) => b.m - a.m); const sortedPaths = partial.map(entry => entry.p); const seconds = timeoutMs % 1000 === 0 ? `${timeoutMs / 1000}` : (timeoutMs / 1000).toFixed(1); - const notice = `glob timed out after ${seconds}s; returning ${sortedPaths.length} partial matches — narrow the pattern instead of retrying blindly`; - return buildResult(sortedPaths, { notice, forceTruncated: true }); + // Walk cost tracks directory-tree size, not pattern specificity: a + // mtime-ranked scan cannot early-exit, so a "narrow" pattern over a + // huge tree still times out. Say so instead of implying the pattern + // was too broad. + const notice = + sortedPaths.length > 0 + ? `glob timed out after ${seconds}s; returning ${sortedPaths.length} partial matches — results are incomplete, scope to a deeper directory instead of retrying blindly` + : `Glob timed out after ${seconds}s before finding any matches — the scan is incomplete, NOT proof of absence. The walk is bounded by directory size, not pattern width; scope the search to a deeper directory (e.g. \`sub/dir/*.ext\` instead of \`*.ext\` at a huge root).`; + return buildResult(sortedPaths, { notice, forceTruncated: true, timedOut: true }); } // Merge per-target results: native glob already ranks each target's own @@ -582,17 +592,20 @@ export const globToolRenderer = { missingPaths.length > 0 ? uiTheme.fg("warning", `skipped missing: ${missingPaths.join(", ")}`) : undefined; if (fileCount === 0) { + // `truncated` on an empty result means the scan timed out mid-walk — + // render "incomplete", not a definitive "No files found". + const emptyLabel = truncated ? "No matches before timeout (scan incomplete)" : "No files found"; const header = renderStatusLine( { icon: "warning", title: "Glob", titleColor: "toolTitle", description: formatGlobRenderPaths(args), - meta: ["0 files"], + meta: truncated ? ["0 files", uiTheme.fg("warning", "timed out")] : ["0 files"], }, uiTheme, ); - const lines = [header, formatEmptyMessage("No files found", uiTheme)]; + const lines = [header, formatEmptyMessage(emptyLabel, uiTheme)]; if (missingNote) lines.push(missingNote); return new Text(lines.join("\n"), 1, 0); } From 295655255a5e5804665fee39cbeb07be74e2d24c Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 03:17:06 +0200 Subject: [PATCH 124/205] test(coding-agent): validated behavioral consistency across core toolset - Added regression tests for browser eval envelope to verify accurate error surfacing and type handling. - Validated glob tool output to distinguish between incomplete timed-out scans and empty results. - Verified exact raw range extraction in read tool to ensure no unwanted context padding is returned. - Confirmed that numbered range reads maintain necessary context padding for readability. --- .../tools/browser-cmux-eval-envelope.test.ts | 52 ++++++++++++++ .../test/tools/glob-renderer.test.ts | 47 ++++++++++++ .../test/tools/read-artifact-large.test.ts | 9 +++ .../test/tools/read-raw-range.test.ts | 72 +++++++++++++++++++ 4 files changed, 180 insertions(+) create mode 100644 packages/coding-agent/test/tools/browser-cmux-eval-envelope.test.ts create mode 100644 packages/coding-agent/test/tools/read-raw-range.test.ts diff --git a/packages/coding-agent/test/tools/browser-cmux-eval-envelope.test.ts b/packages/coding-agent/test/tools/browser-cmux-eval-envelope.test.ts new file mode 100644 index 000000000..0b4f1bb3f --- /dev/null +++ b/packages/coding-agent/test/tools/browser-cmux-eval-envelope.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "bun:test"; +import { serializeEvalWithEnvelope, unwrapEvalEnvelope } from "@oh-my-pi/pi-coding-agent/tools/browser/cmux/rpc"; + +/** + * Executes the envelope script the way the cmux daemon would (global-scope + * evaluation of an expression) and round-trips the result through JSON to + * mirror the socket wire format. + */ +function runOnWire(script: string): unknown { + const value = new Function(`return (${script})`)(); + return JSON.parse(JSON.stringify(value)); +} + +describe("cmux eval envelope", () => { + it("returns plain values through the ok envelope", () => { + const value = runOnWire(serializeEvalWithEnvelope("1 + 1", [])); + expect(unwrapEvalEnvelope(value, "tab.evaluate()")).toBe(2); + }); + + it("invokes function sources with serialized args", () => { + const script = serializeEvalWithEnvelope( + ((a: number, b: number) => a * b) as (...args: unknown[]) => unknown, + [6, 7], + ); + const value = runOnWire(script); + expect(unwrapEvalEnvelope(value, "tab.evaluate()")).toBe(42); + }); + + it("surfaces thrown exceptions with their message instead of an opaque js_error", () => { + // Regression: a throwing script came back as the daemon's bare + // `js_error: A JavaScript exception occurred`, hiding the actual error. + const script = serializeEvalWithEnvelope("(() => { throw new Error('boom from page') })()", []); + const value = runOnWire(script); + expect(() => unwrapEvalEnvelope(value, "tab.evaluate()")).toThrow(/boom from page/); + }); + + it("flags Promise returns with an actionable error instead of an unsupported-type failure", () => { + const script = serializeEvalWithEnvelope("Promise.resolve(1)", []); + const value = runOnWire(script); + expect(() => unwrapEvalEnvelope(value, "tab.evaluate()")).toThrow(/synchronously/); + }); + + it("maps undefined results to null (JSON cannot carry undefined)", () => { + const value = runOnWire(serializeEvalWithEnvelope("undefined", [])); + expect(unwrapEvalEnvelope(value, "tab.evaluate()")).toBeNull(); + }); + + it("passes through values from daemons that did not run the wrapper", () => { + expect(unwrapEvalEnvelope<{ plain: boolean }>({ plain: true }, "tab.evaluate()")).toEqual({ plain: true }); + expect(unwrapEvalEnvelope(7, "tab.evaluate()")).toBe(7); + }); +}); diff --git a/packages/coding-agent/test/tools/glob-renderer.test.ts b/packages/coding-agent/test/tools/glob-renderer.test.ts index 1b804529a..fe8dbba82 100644 --- a/packages/coding-agent/test/tools/glob-renderer.test.ts +++ b/packages/coding-agent/test/tools/glob-renderer.test.ts @@ -25,4 +25,51 @@ describe("globToolRenderer", () => { expect(renderedLines[0]).not.toContain(uiTheme.fg("accent", uiTheme.symbol("icon.search"))); expect(renderedLines[0]).not.toContain(uiTheme.fg("accent", "Find")); }); + + it("renders a timed-out empty scan as incomplete instead of a definitive no-files claim", async () => { + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + const uiTheme = theme!; + // `truncated` with zero files only happens on the timeout path — the + // scan died mid-walk, so "No files found" would be a false claim. + const result = { + content: [{ type: "text", text: "Glob timed out after 5s before finding any matches" }], + details: { + fileCount: 0, + files: [], + truncated: true, + }, + }; + + const renderedLines = globToolRenderer + .renderResult(result as never, { expanded: true, isPartial: false }, uiTheme, { paths: "~/.cache/*" }) + .render(240); + const plain = sanitizeText(renderedLines.join("\n")); + + expect(plain).toContain("No matches before timeout (scan incomplete)"); + expect(plain).toContain("timed out"); + expect(plain).not.toContain("No files found"); + }); + + it("renders a genuinely empty result as no files found", async () => { + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + const uiTheme = theme!; + const result = { + content: [{ type: "text", text: "No files found matching pattern" }], + details: { + fileCount: 0, + files: [], + truncated: false, + }, + }; + + const renderedLines = globToolRenderer + .renderResult(result as never, { expanded: true, isPartial: false }, uiTheme, { paths: "src/*.zig" }) + .render(240); + const plain = sanitizeText(renderedLines.join("\n")); + + expect(plain).toContain("No files found"); + expect(plain).not.toContain("incomplete"); + }); }); diff --git a/packages/coding-agent/test/tools/read-artifact-large.test.ts b/packages/coding-agent/test/tools/read-artifact-large.test.ts index 730e971d9..46831a30f 100644 --- a/packages/coding-agent/test/tools/read-artifact-large.test.ts +++ b/packages/coding-agent/test/tools/read-artifact-large.test.ts @@ -95,6 +95,15 @@ describe("read tool large artifact handling", () => { expect(output).not.toContain("artifact://0:raw:N-M"); }); + it("returns exactly the requested raw artifact range without context padding", async () => { + const result = await tool.execute("call-raw-exact", { path: "artifact://0:raw:31-31" }); + const output = getTextOutput(result); + + expect(output).toContain("line-031"); + expect(output).not.toContain("line-030"); + expect(output).not.toContain("line-032"); + }); + it("shortens artifact paths under the user's home dir instead of leaking the absolute path", async () => { const homeSpy = spyOn(os, "homedir").mockReturnValue(testDir); try { diff --git a/packages/coding-agent/test/tools/read-raw-range.test.ts b/packages/coding-agent/test/tools/read-raw-range.test.ts new file mode 100644 index 000000000..75d5c3cc9 --- /dev/null +++ b/packages/coding-agent/test/tools/read-raw-range.test.ts @@ -0,0 +1,72 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; + +function getTextOutput(result: { content: Array<{ type: string; text?: string }> }): string { + return result.content + .filter(c => c.type === "text" && typeof c.text === "string") + .map(c => c.text as string) + .join("\n"); +} + +function makeSession(cwd: string): ToolSession { + return { + cwd, + hasUI: false, + getSessionFile: () => path.join(cwd, "session.jsonl"), + getSessionSpawns: () => "*", + getArtifactsDir: () => path.join(cwd, "session"), + settings: Settings.isolated(), + }; +} + +describe("read tool raw range exactness", () => { + let testDir: string; + let filePath: string; + let tool: ReadTool; + + beforeEach(async () => { + testDir = await fs.mkdtemp(path.join(os.tmpdir(), "read-raw-range-")); + filePath = path.join(testDir, "data.txt"); + const lines = Array.from({ length: 60 }, (_, index) => `L${String(index + 1).padStart(2, "0")}`); + await Bun.write(filePath, lines.join("\n")); + tool = new ReadTool(makeSession(testDir)); + }); + + afterEach(async () => { + await fs.rm(testDir, { recursive: true, force: true }); + }); + + it("returns exactly the requested single line for raw:N-N", async () => { + // Regression: raw ranges used to get 1 leading + 3 trailing context + // lines. Without line numbers the padding is indistinguishable from + // requested content, so verbatim-extraction callers pasted 5 lines + // where they asked for 1. + const result = await tool.execute("call-raw-single", { path: `${filePath}:raw:31-31` }); + const output = getTextOutput(result); + + expect(output.trimEnd()).toBe("L31"); + }); + + it("returns exactly the requested raw range at the start of the file", async () => { + const result = await tool.execute("call-raw-head", { path: `${filePath}:raw:1-2` }); + const output = getTextOutput(result); + + expect(output.trimEnd()).toBe("L01\nL02"); + }); + + it("keeps context padding for numbered range reads", async () => { + // Numbered mode intentionally pads (leading anchor buffer + trailing + // disambiguation lines) — line numbers make the padding self-describing. + const result = await tool.execute("call-numbered", { path: `${filePath}:31-31` }); + const output = getTextOutput(result); + + expect(output).toContain("L31"); + expect(output).toContain("L30"); + expect(output).toContain("L32"); + }); +}); From 31c9f48509509fb5dfe758e9e27972e9caffe525 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 03:17:05 +0200 Subject: [PATCH 125/205] fix(ai): prevented inclusion of empty image placeholders in tool outputs - Stopped serializing "(see attached image)" for genuinely empty tool results. - Added logic to verify presence of actual image content before emitting placeholders. - Verified fix with new regression tests ensuring empty content remains empty. --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/openai-shared.ts | 9 +- ...openai-responses-empty-tool-result.test.ts | 89 +++++++++++++++++++ 3 files changed, 101 insertions(+), 1 deletion(-) create mode 100644 packages/ai/test/openai-responses-empty-tool-result.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 1a0a75a8d..287dc5f55 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the Responses API serializing a genuinely empty tool result (e.g. reading an empty file with `:raw`) as `(see attached image)` even when the turn carried no image, sending models chasing a phantom attachment. The placeholder is now emitted only when the result actually contains images; empty results stay empty, matching the Completions and Google converters. + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 75e8eeaa1..55145b610 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1720,12 +1720,19 @@ export function appendResponsesToolResultMessages( const hasImages = toolResult.content.some((block): block is ImageContent => block.type === "image"); const omittedImages = hasImages && !supportsImages; const normalized = normalizeResponsesToolCallId(toolResult.toolCallId); + // "(see attached image)" is only truthful when the result actually carries + // images (they ride as a separate user message on the Responses API). A + // genuinely empty text result (empty file read, silent tool) must stay + // empty — the placeholder sent models chasing an attachment that never + // existed. const output = ( omittedImages ? joinTextWithImagePlaceholder(textResult, true) : textResult.length > 0 ? textResult - : "(see attached image)" + : hasImages + ? "(see attached image)" + : "" ).toWellFormed(); if (strictResponsesPairing && !knownCallIds.has(normalized.callId)) { // Strict backends (Azure, Copilot) reject unpaired outputs outright, but diff --git a/packages/ai/test/openai-responses-empty-tool-result.test.ts b/packages/ai/test/openai-responses-empty-tool-result.test.ts new file mode 100644 index 000000000..99faae05f --- /dev/null +++ b/packages/ai/test/openai-responses-empty-tool-result.test.ts @@ -0,0 +1,89 @@ +import { describe, expect, it } from "bun:test"; +import { buildResponsesInput } from "@oh-my-pi/pi-ai/providers/openai-shared"; +import type { Context, ImageContent, ModelSpec, TextContent } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const model = buildModel({ + id: "test-vision", + name: "Test Vision", + api: "openai-responses", + provider: "openai", + baseUrl: "https://api.openai.com/v1", + reasoning: false, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128000, + maxTokens: 16000, +} satisfies ModelSpec<"openai-responses">); + +const zeroUsage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function makeContext(content: (TextContent | ImageContent)[]): Context { + return { + messages: [ + { + role: "assistant", + content: [{ type: "toolCall", id: "call_1", name: "read", arguments: { path: "empty.txt" } }], + api: "openai-responses", + provider: "openai", + model: "test-vision", + usage: zeroUsage, + stopReason: "toolUse", + timestamp: Date.now(), + }, + { + role: "toolResult", + toolCallId: "call_1", + toolName: "read", + content, + isError: false, + timestamp: Date.now(), + }, + ], + }; +} + +function findFunctionCallOutput(items: unknown[]): string | undefined { + for (const item of items) { + if (!item || typeof item !== "object") continue; + if (!("type" in item) || item.type !== "function_call_output") continue; + if ("output" in item && typeof item.output === "string") return item.output; + } + return undefined; +} + +describe("Responses API empty tool result", () => { + it("keeps a genuinely empty text result empty instead of claiming an attached image", () => { + // Regression: an empty tool result (e.g. reading an empty file with + // `:raw`) was serialized as "(see attached image)" with no image + // anywhere in the turn, sending models chasing a phantom attachment. + const items = buildResponsesInput({ + model, + context: makeContext([{ type: "text", text: "" }]), + strictResponsesPairing: true, + supportsImageDetailOriginal: true, + }); + + expect(findFunctionCallOutput(items)).toBe(""); + }); + + it("keeps the placeholder when the result actually carries an image", () => { + // Images ride as a separate user message on the Responses API; the + // function output must point the model at them. + const items = buildResponsesInput({ + model, + context: makeContext([{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" }]), + strictResponsesPairing: true, + supportsImageDetailOriginal: true, + }); + + expect(findFunctionCallOutput(items)).toBe("(see attached image)"); + }); +}); From 54af1c03fd80af0b9cf3ab7c3b827030ed3b6e7d Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 05:00:02 +0200 Subject: [PATCH 126/205] fix(coding-agent): ensured provider errors are surfaced to ACP clients - Added a fallback to emit error messages during `agent_end` if no error was previously streamed during the turn. - Added tracking to prevent duplicate error messages when a provider error is successfully delivered during streaming. - Added a stderr hint for interactive users launching the ACP server directly from a terminal. --- packages/coding-agent/CHANGELOG.md | 7 ++ .../coding-agent/src/modes/acp/acp-agent.ts | 66 ++++++++++++- .../src/modes/acp/acp-event-mapper.ts | 5 + .../coding-agent/src/modes/acp/acp-mode.ts | 11 +++ packages/coding-agent/test/acp-agent.test.ts | 96 +++++++++++++++++++ 5 files changed, 184 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 52ff045e4..a3d94abdb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- `omp acp` now prints a short hint on stderr when launched from an interactive terminal (stdin is a TTY): the command speaks JSON-RPC over stdout and is meant to be spawned by an ACP client such as Zed, so running it by hand previously showed nothing at all + ### Changed - Refined agent delegation logic to prioritize top-level planning and scoping by the primary agent @@ -15,7 +19,10 @@ ### Fixed +- Fixed silent failures in ACP mode when provider errors occurred before streaming assistant text +- Prevented duplicate error messages in ACP when a provider error was both streamed and final - Fixed `glob` reporting the contradictory "No files found matching pattern" next to a "timed out; returning 0 partial matches" notice. A timed-out empty scan now states explicitly that the result is incomplete (not proof of absence) and suggests scoping to a deeper directory, and the TUI renders it as "No matches before timeout (scan incomplete)" instead of a definitive no-files claim. +- Fixed ACP turns ending silently when the provider request failed before streaming any assistant output (e.g. GitHub Copilot's intermittent `HTTP 400 model_not_supported` after retries): such failures emit only `agent_end` with the error on the assistant message, which never mapped to a session update. The error message is now delivered to the client as an `agent_message_chunk`, without duplicating errors that already streamed - Fixed `read` adding invisible context padding to raw range selectors: `raw:31-31` returned lines 30–34 (1 leading + 3 trailing context lines) with nothing to distinguish the padding, corrupting verbatim-extraction workflows. Raw ranges now return exactly the requested lines across all three range paths (plain files, in-memory/bridge reads, artifacts); numbered reads keep the self-describing padding. - Fixed browser screenshots capturing the wrong tab or hanging until the op timeout when multiple tabs share one Chromium (sibling headless tabs, `cdp_url`/app attach): CDP reads the *active* target's compositor surface, so a backgrounded page could stall waiting for a frame or return a sibling's pixels. The worker now activates the page (`bringToFront`) before every capture, best-effort. - Fixed cmux `tab.screenshot({ selector })` silently returning a full-viewport capture that models consumed as an element crop. The cmux daemon has no element-clip or full-page capture; the tool still scrolls the selector into view but now labels the image as full-viewport (same for `fullPage`) instead of mislabeling it. diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 6d76f2721..5b31ec935 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -129,6 +129,15 @@ type PromptLifecycleError = Error & { readonly code: "ACP_SESSION_CLOSED" }; type PromptTurnState = { cancelRequested: boolean; settled: boolean; + /** + * Delivery of streamed assistant `error` chunks this turn (the mapper + * surfaces them as `agent_message_chunk`s). Resolves `true` once at least + * one error chunk reached the client — the `agent_end` error fallback in + * {@link AcpAgent##flushUnreportedTurnError} awaits it and stays silent on + * success, so a fallback racing an in-flight delivery can neither duplicate + * the error nor drop it when delivery fails. + */ + errorTextDelivery: Promise | undefined; /** * `abort()` is in-flight (or its bounded-timeout race). `undefined` while the turn is * running normally and after cleanup completes. The turn occupies `record.promptTurn` @@ -684,6 +693,7 @@ export class AcpAgent implements Agent { record.promptTurn = { cancelRequested: false, settled: false, + errorTextDelivery: undefined, cleanup: undefined, usageBaseline: this.#cloneUsageStatistics(record.session.sessionManager.getUsageStatistics()), unsubscribe: undefined, @@ -1200,6 +1210,10 @@ export class AcpAgent implements Agent { imageDataCache.set(key, resolved); return resolved; }; + const streamedAssistantError = + event.type === "message_update" && + event.message.role === "assistant" && + event.assistantMessageEvent.type === "error"; for (const notification of mapAgentSessionEventToAcpSessionUpdates(event, record.session.sessionId, { getMessageId: message => this.#getLiveMessageId(record, message), getMessageProgress: message => this.#getLiveMessageProgress(record, message), @@ -1207,7 +1221,18 @@ export class AcpAgent implements Agent { cwd: record.session.sessionManager.getCwd(), resolveImageData: resolveImageDataForAcp, })) { - await this.#connection.sessionUpdate(notification); + const delivery = this.#connection.sessionUpdate(notification); + if (streamedAssistantError) { + // Resolves true only once the error chunk actually reached the + // client — a failed delivery keeps the agent_end fallback armed. + const outcome = delivery.then( + () => true, + () => false, + ); + const prior = promptTurn.errorTextDelivery; + promptTurn.errorTextDelivery = prior ? Promise.all([prior, outcome]).then(([a, b]) => a || b) : outcome; + } + await delivery; } if (event.type === "tool_execution_end") { record.toolArgsById.delete(event.toolCallId); @@ -1216,6 +1241,7 @@ export class AcpAgent implements Agent { if (event.type === "agent_end") { await this.#flushMissedFinalAssistantText(record, event); + await this.#flushUnreportedTurnError(record, event); await this.#emitEndOfTurnUpdates(record); await this.#waitForAcpPromptIdle(record); record.liveMessageId = undefined; @@ -1272,6 +1298,44 @@ export class AcpAgent implements Agent { }); } + /** + * Surface a turn-fatal provider error that never reached the client. A + * request that fails before streaming any assistant events — e.g. GitHub + * Copilot's `HTTP 400 model_not_supported` after retries — emits only + * `agent_end` with an empty assistant message carrying `errorMessage` + * (`Agent#runLoop`'s catch), so no `message_update`/`message_end` ever maps + * to a session update and the client sees the turn end silently. Errors + * that did stream are tracked via {@link PromptTurnState.errorTextDelivery}; + * the fallback awaits that delivery and re-sends only when it failed. + */ + async #flushUnreportedTurnError( + record: ManagedSessionRecord, + event: Extract, + ): Promise { + const streamedDelivery = record.promptTurn?.errorTextDelivery; + if (streamedDelivery && (await streamedDelivery)) { + return; + } + const lastAssistant = [...event.messages] + .reverse() + .find((message): message is AssistantMessage => message.role === "assistant"); + if (lastAssistant?.stopReason !== "error") { + return; + } + const errorMessage = lastAssistant.errorMessage; + if (!errorMessage || isSilentAbort(lastAssistant)) { + return; + } + await this.#connection.sessionUpdate({ + sessionId: record.session.sessionId, + update: { + sessionUpdate: "agent_message_chunk", + content: { type: "text", text: errorMessage }, + messageId: record.liveMessageId ?? crypto.randomUUID(), + }, + }); + } + async #waitForAcpPromptIdle(record: ManagedSessionRecord): Promise { for (let pass = 0; pass < ACP_ASYNC_DELIVERY_DRAIN_MAX_PASSES; pass++) { await record.session.waitForIdle(); diff --git a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts index 57f0cd388..f3536b598 100644 --- a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts +++ b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts @@ -284,6 +284,11 @@ function mapAssistantMessageUpdate( case "error": sessionUpdate = "agent_message_chunk"; text = event.assistantMessageEvent.error.errorMessage ?? "Unknown error"; + // The surfaced error is the message's visible text: keeps the + // message_end / agent_end fallbacks from emitting again. + if (text.length > 0 && progress) { + progress.textEmitted = true; + } break; default: return []; diff --git a/packages/coding-agent/src/modes/acp/acp-mode.ts b/packages/coding-agent/src/modes/acp/acp-mode.ts index 8aaf05409..690d5f4e2 100644 --- a/packages/coding-agent/src/modes/acp/acp-mode.ts +++ b/packages/coding-agent/src/modes/acp/acp-mode.ts @@ -14,6 +14,17 @@ export function createAcpConnection( } export async function runAcpMode(createSession: AcpSessionFactory, initialSession?: AgentSession): Promise { + // Humans who run `omp acp` by hand see a silent process and assume it is + // broken (stdout is the JSON-RPC transport, so nothing may be printed + // there). When stdin is a TTY no ACP client is attached — say so on stderr + // before the transport starts. + if (process.stdin.isTTY) { + process.stderr.write( + "omp acp: ACP server speaking JSON-RPC over stdio.\n" + + 'This command is meant to be spawned by an ACP client (e.g. Zed\'s "agent_servers" config), not run directly.\n' + + "Waiting for protocol frames on stdin; logs: ~/.omp/logs/\n", + ); + } const input = stream.Writable.toWeb(process.stdout); const output = stream.Readable.toWeb(process.stdin); const transport = ndJsonStream(input, output); diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index 7e9554119..f8892c557 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -1127,6 +1127,102 @@ describe("ACP agent", () => { await Bun.sleep(0); }); + it("surfaces a provider error that reaches the client only via agent_end", async () => { + // A request that fails before streaming any assistant events (e.g. + // GitHub Copilot's HTTP 400 model_not_supported after retries) emits no + // message_update/message_end — only agent_end carrying an empty + // assistant message with errorMessage. The client must still see why + // the turn ended instead of a silent stop. + const harness = await createHarness(); + const created = await harness.agent.newSession({ cwd: harness.cwdA, mcpServers: [] }); + const session = harness.findSession(created.sessionId); + if (!session) throw new Error("session not registered"); + + const errorText = + "GitHub Copilot rejected this model (HTTP 400 model_not_supported) after retries. Try again in a few seconds."; + const failedMessage = { + ...makeAssistantMessage(""), + stopReason: "error" as const, + errorMessage: errorText, + }; + session.prompt = async (text: string): Promise => { + session.promptCalls.push(text); + session.isStreaming = true; + session.sessionManager.appendMessage(failedMessage); + for (const listener of session.listeners()) { + listener({ type: "agent_end", messages: [failedMessage] } as AgentSessionEvent); + } + session.isStreaming = false; + return true; + }; + + const response = await harness.agent.prompt({ + sessionId: created.sessionId, + prompt: [{ type: "text", text: "Say hello" }], + }); + expectAcpStructure(zPromptResponse, response); + + const messageChunks = harness.updates.filter( + update => update.sessionId === created.sessionId && update.update.sessionUpdate === "agent_message_chunk", + ); + expect(messageChunks).toHaveLength(1); + expect(messageChunks[0]?.update).toEqual(expect.objectContaining({ content: { type: "text", text: errorText } })); + expectAcpNotifications(harness.updates); + + harness.abortController.abort(); + await Bun.sleep(0); + }); + + it("does not re-send a streamed error chunk from the agent_end fallback", async () => { + // When the error DID stream (message_update with an `error` event maps + // to an agent_message_chunk), the agent_end fallback must stay silent — + // even though agent_end races the in-flight chunk delivery. + const harness = await createHarness(); + const created = await harness.agent.newSession({ cwd: harness.cwdA, mcpServers: [] }); + const session = harness.findSession(created.sessionId); + if (!session) throw new Error("session not registered"); + + const errorText = "upstream stream failed"; + const failedMessage = { + ...makeAssistantMessage(""), + stopReason: "error" as const, + errorMessage: errorText, + }; + session.prompt = async (text: string): Promise => { + session.promptCalls.push(text); + session.isStreaming = true; + for (const listener of session.listeners()) { + listener({ + type: "message_update", + message: failedMessage, + assistantMessageEvent: { type: "error", error: { errorMessage: errorText } }, + } as AgentSessionEvent); + } + session.sessionManager.appendMessage(failedMessage); + for (const listener of session.listeners()) { + listener({ type: "agent_end", messages: [failedMessage] } as AgentSessionEvent); + } + session.isStreaming = false; + return true; + }; + + const response = await harness.agent.prompt({ + sessionId: created.sessionId, + prompt: [{ type: "text", text: "Say hello" }], + }); + expectAcpStructure(zPromptResponse, response); + + const messageChunks = harness.updates.filter( + update => update.sessionId === created.sessionId && update.update.sessionUpdate === "agent_message_chunk", + ); + expect(messageChunks).toHaveLength(1); + expect(messageChunks[0]?.update).toEqual(expect.objectContaining({ content: { type: "text", text: errorText } })); + expectAcpNotifications(harness.updates); + + harness.abortController.abort(); + await Bun.sleep(0); + }); + it("replays assistant tool calls and matching results without duplicating the start", async () => { const harness = await createHarness(); const stored = new FakeAgentSession(harness.cwdA); From 75bac085a744e6671dceeae52722fa64c2f64b7e Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 02:12:07 +0200 Subject: [PATCH 127/205] feat(vibe): implemented persistent worker session infrastructure - Introduced a registry and runtime for managing persistent worker subagent sessions. - Added lifecycle management capabilities including spawning, dispatching, waiting, and terminating background jobs. - Implemented TUI visualization tools to track and render worker session states. - Enabled subagent session continuation via follow-up turn processing. --- packages/coding-agent/src/task/executor.ts | 105 ++++ packages/coding-agent/src/tools/vibe.ts | 364 ++++++++++++ packages/coding-agent/src/vibe/runtime.ts | 635 +++++++++++++++++++++ packages/coding-agent/src/vibe/state.ts | 4 + 4 files changed, 1108 insertions(+) create mode 100644 packages/coding-agent/src/tools/vibe.ts create mode 100644 packages/coding-agent/src/vibe/runtime.ts create mode 100644 packages/coding-agent/src/vibe/state.ts diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index a640b658d..0b24afa55 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -1880,6 +1880,111 @@ export async function finalizeSubagentLifecycle(args: { }); } +/** Options for {@link runSubagentFollowUpTurn}. */ +export interface FollowUpTurnOptions { + /** Registry id of the (live or parked) subagent to continue. */ + id: string; + /** Agent definition the session was originally spawned with (drives progress labels + finalize). */ + agent: AgentDefinition; + /** The follow-up message; sent as the turn's user prompt. */ + message: string; + index?: number; + description?: string; + signal?: AbortSignal; + onProgress?: (progress: AgentProgress) => void; + eventBus?: EventBus; + parentToolCallId?: string; + /** When set, the turn's raw output is (re)written to `/.md` so `agent://` tracks the latest turn. */ + artifactsDir?: string; + /** Wall-clock cap in ms for this turn; 0 disables. */ + maxRuntimeMs?: number; +} + +/** + * Continue a previously spawned (keep-alive) subagent with one more monitored + * turn: revive it if parked, send `message` as a real prompt, drive it to + * `yield`, and finalize a {@link SingleResult} exactly like a first run. + * + * The session's full conversation history is retained (live session, or JSONL + * replay through the lifecycle reviver), so the turn sees all prior context. + * Unlike {@link runSubprocess}, the session is NOT torn down afterwards — it + * stays adopted by the {@link AgentLifecycleManager} (idle → TTL park → + * revive), and an aborted turn only aborts the in-flight turn. + */ +export async function runSubagentFollowUpTurn(options: FollowUpTurnOptions): Promise { + const { id, agent, message, signal } = options; + const index = options.index ?? 0; + const startTime = Date.now(); + const session = await AgentLifecycleManager.global().ensureLive(id); + const ref = AgentRegistry.global().get(id); + const sessionFile = ref?.sessionFile ?? undefined; + + const monitor = createSubagentRunMonitor({ + index, + id, + agent, + task: message, + description: options.description, + signal, + onProgress: options.onProgress, + eventBus: options.eventBus, + parentToolCallId: options.parentToolCallId, + detached: true, + sessionFile, + softRequestBudget: 0, + softRequestBudgetNotice: false, + maxRuntimeMs: options.maxRuntimeMs ?? 0, + }); + + if (options.eventBus) { + options.eventBus.emit(TASK_SUBAGENT_LIFECYCLE_CHANNEL, { + id, + agent: agent.name, + parentToolCallId: options.parentToolCallId, + detached: true, + agentSource: agent.source, + description: options.description, + status: "started", + sessionFile, + index, + }); + } + + monitor.setActiveSession(session); + const unsubscribe = monitor.attach(session); + let outcome: DriveOutcome; + try { + outcome = await driveSessionToYield(session, monitor, message); + } finally { + try { + await untilAborted(AbortSignal.timeout(5000), () => monitor.waitForActiveSessionAbort()); + } catch { + // Ignore abort cleanup timeouts; the session stays adopted either way. + } + unsubscribe(); + const active = monitor.takeActiveSession(); + if (active) monitor.captureSalvage(active); + monitor.finish(); + } + + return finalizeRunResult({ + monitor, + done: { ...outcome, abortReason: outcome.abortReasonText, durationMs: Date.now() - startTime }, + index, + id, + agent, + task: message, + description: options.description, + signal, + artifactsDir: options.artifactsDir, + eventBus: options.eventBus, + parentToolCallId: options.parentToolCallId, + detached: true, + sessionFile, + startTime, + }); +} + /** * Run a single agent in-process. */ diff --git a/packages/coding-agent/src/tools/vibe.ts b/packages/coding-agent/src/tools/vibe.ts new file mode 100644 index 000000000..8b29a8beb --- /dev/null +++ b/packages/coding-agent/src/tools/vibe.ts @@ -0,0 +1,364 @@ +/** + * Vibe mode tools — the director's entire non-read surface. + * + * Five thin tools over {@link VibeSessionRegistry}: spawn/send/wait/kill/list + * persistent worker sessions ("fast"/"good" CLIs). Spawns and sends return + * immediately; turn results self-deliver through the async job manager. + */ +import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import type { Component } from "@oh-my-pi/pi-tui"; +import { Text } from "@oh-my-pi/pi-tui"; +import { prompt } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; +import type { RenderResultOptions } from "../extensibility/custom-tools/types"; +import type { Theme } from "../modes/theme/theme"; +import vibeKillDescription from "../prompts/tools/vibe-kill.md" with { type: "text" }; +import vibeListDescription from "../prompts/tools/vibe-list.md" with { type: "text" }; +import vibeSendDescription from "../prompts/tools/vibe-send.md" with { type: "text" }; +import vibeSpawnDescription from "../prompts/tools/vibe-spawn.md" with { type: "text" }; +import vibeWaitDescription from "../prompts/tools/vibe-wait.md" with { type: "text" }; +import { MAIN_AGENT_ID } from "../registry/agent-registry"; +import { oneLineLabel } from "../task/types"; +import { renderStatusLine } from "../tui"; +import { + type VibeCli, + type VibeKillOutcome, + type VibeRosterEntry, + type VibeSendOutcome, + VibeSessionRegistry, + type VibeSessionState, +} from "../vibe/runtime"; +import type { ToolSession } from "./index"; +import { + Ellipsis, + formatBadge, + formatStatusIcon, + replaceTabs, + type ToolUIColor, + type ToolUIStatus, + truncateToWidth, +} from "./render-utils"; + +export const VIBE_TOOL_NAMES = ["vibe_spawn", "vibe_send", "vibe_wait", "vibe_kill", "vibe_list"] as const; + +const vibeSpawnSchema = type({ + cli: type("'fast' | 'good'").describe( + "worker flavor: fast = low-latency model for mechanical work; good = strong model for hard work", + ), + "name?": type("string <= 48").describe("optional session name; generated when omitted"), + prompt: type("string > 0").describe("first instruction; the worker starts with no other context"), +}); + +const vibeSendSchema = type({ + session: type("string > 0").describe("session id from vibe_spawn / vibe_list"), + message: type("string > 0").describe("message for the session; steers mid-turn, else runs as its next turn"), +}); + +const vibeWaitSchema = type({ + "sessions?": type("string[]").describe("session ids to watch; omit to watch every session with a turn in flight"), + "timeout?": type("number > 0").describe("max seconds to wait (default 30)"), +}); + +const vibeKillSchema = type({ + session: type("string > 0").describe("session id to terminate"), +}); + +const vibeListSchema = type({}); + +type VibeOp = "spawn" | "send" | "wait" | "kill" | "list"; + +/** Details payload shared by every vibe tool for TUI rendering. */ +export interface VibeToolDetails { + op: VibeOp; + roster: VibeRosterEntry[]; + spawned?: { id: string; cli: VibeCli; jobId: string }; + send?: VibeSendOutcome; + wait?: { + settled: Array<{ id: string; jobId: string; status: "completed" | "failed" | "cancelled" }>; + stillRunning: string[]; + timedOut: boolean; + }; + killed?: VibeKillOutcome; +} + +function rosterOf(session: ToolSession): VibeRosterEntry[] { + return VibeSessionRegistry.global().list(session.getAgentId?.() ?? MAIN_AGENT_ID); +} + +function textResult(text: string, details: VibeToolDetails): AgentToolResult { + return { content: [{ type: "text", text }], details }; +} + +export class VibeSpawnTool implements AgentTool { + readonly name = "vibe_spawn"; + readonly approval = "exec" as const; + readonly label = "Vibe Spawn"; + readonly summary = "Start a persistent fast/good worker session"; + readonly description: string; + readonly parameters = vibeSpawnSchema; + readonly strict = true; + constructor(private readonly session: ToolSession) { + this.description = prompt.render(vibeSpawnDescription); + } + + async execute(_toolCallId: string, params: typeof vibeSpawnSchema.infer): Promise> { + const { id, jobId } = await VibeSessionRegistry.global().spawn(this.session, params); + return textResult( + `Spawned ${params.cli} session \`${id}\` (turn job \`${jobId}\`). The turn result will be delivered when it finishes — keep directing other sessions meanwhile. Continue this one with vibe_send \`${id}\`.`, + { op: "spawn", roster: rosterOf(this.session), spawned: { id, cli: params.cli, jobId } }, + ); + } +} + +export class VibeSendTool implements AgentTool { + readonly name = "vibe_send"; + readonly approval = "exec" as const; + readonly label = "Vibe Send"; + readonly summary = "Message a worker session (steer or next turn)"; + readonly description: string; + readonly parameters = vibeSendSchema; + readonly strict = true; + constructor(private readonly session: ToolSession) { + this.description = prompt.render(vibeSendDescription); + } + + async execute(_toolCallId: string, params: typeof vibeSendSchema.infer): Promise> { + const outcome = await VibeSessionRegistry.global().send(this.session, params); + const ack = + outcome.mode === "turn" + ? `Started a new turn on \`${outcome.id}\` (job \`${outcome.jobId}\`). Its result will be delivered when the turn finishes.` + : outcome.mode === "steered" + ? `Steered \`${outcome.id}\` mid-turn — the running turn sees your message at its next step.` + : `\`${outcome.id}\` is mid-turn; your message is queued and runs automatically as the next turn.`; + return textResult(ack, { op: "send", roster: rosterOf(this.session), send: outcome }); + } +} + +export class VibeWaitTool implements AgentTool { + readonly name = "vibe_wait"; + readonly approval = "read" as const; + readonly label = "Vibe Wait"; + readonly summary = "Block until a worker session finishes its turn"; + readonly description: string; + readonly parameters = vibeWaitSchema; + readonly strict = true; + readonly interruptible = true; + constructor(private readonly session: ToolSession) { + this.description = prompt.render(vibeWaitDescription); + } + + async execute( + _toolCallId: string, + params: typeof vibeWaitSchema.infer, + signal?: AbortSignal, + ): Promise> { + const outcome = await VibeSessionRegistry.global().wait(this.session, { + sessions: params.sessions, + timeoutMs: params.timeout !== undefined ? params.timeout * 1000 : undefined, + signal, + }); + const details: VibeToolDetails = { + op: "wait", + roster: rosterOf(this.session), + wait: { + settled: outcome.settled.map(({ id, jobId, status }) => ({ id, jobId, status })), + stillRunning: outcome.stillRunning, + timedOut: outcome.timedOut, + }, + }; + if (outcome.settled.length === 0 && outcome.stillRunning.length === 0) { + return { ...textResult("No turns in flight to wait for.", details), useless: true }; + } + const lines: string[] = []; + for (const entry of outcome.settled) { + lines.push(`## \`${entry.id}\` — ${entry.status}`, entry.resultText, ""); + } + if (outcome.stillRunning.length > 0) { + lines.push(`Still running: ${outcome.stillRunning.map(id => `\`${id}\``).join(", ")}.`); + } + if (outcome.timedOut) { + lines.push("Wait window elapsed before any turn settled — re-issue vibe_wait to keep waiting."); + } + const result = textResult(lines.join("\n").trimEnd(), details); + // A pure "still waiting" frame is noise once a newer wait exists. + return outcome.settled.length === 0 ? { ...result, useless: true } : result; + } +} + +export class VibeKillTool implements AgentTool { + readonly name = "vibe_kill"; + readonly approval = "read" as const; + readonly label = "Vibe Kill"; + readonly summary = "Terminate a worker session"; + readonly description: string; + readonly parameters = vibeKillSchema; + readonly strict = true; + constructor(private readonly session: ToolSession) { + this.description = prompt.render(vibeKillDescription); + } + + async execute(_toolCallId: string, params: typeof vibeKillSchema.infer): Promise> { + const outcome = await VibeSessionRegistry.global().kill(this.session, params.session); + const cancelNote = outcome.cancelledTurn ? " Its in-flight turn was cancelled." : ""; + return textResult( + `Killed session \`${outcome.id}\`.${cancelNote} Transcript remains at history://${outcome.id}.`, + { op: "kill", roster: rosterOf(this.session), killed: outcome }, + ); + } +} + +export class VibeListTool implements AgentTool { + readonly name = "vibe_list"; + readonly approval = "read" as const; + readonly label = "Vibe List"; + readonly summary = "List worker sessions and their states"; + readonly description: string; + readonly parameters = vibeListSchema; + readonly strict = true; + constructor(private readonly session: ToolSession) { + this.description = prompt.render(vibeListDescription); + } + + async execute(): Promise> { + const roster = rosterOf(this.session); + const details: VibeToolDetails = { op: "list", roster }; + if (roster.length === 0) { + return textResult("No vibe sessions. Spawn one with vibe_spawn.", details); + } + const lines = roster.map(entry => { + const parts = [ + `- \`${entry.id}\` [${entry.cli}] ${entry.state}`, + `${entry.turns} turn${entry.turns === 1 ? "" : "s"}`, + ]; + if (entry.queued > 0) parts.push(`${entry.queued} queued`); + if (entry.model) parts.push(entry.model); + if (entry.lastActivity) parts.push(`last: ${entry.lastActivity}`); + return parts.join(" · "); + }); + return textResult(lines.join("\n"), details); + } +} + +// ============================================================================= +// TUI Renderer +// ============================================================================= + +const ROSTER_LINE_WIDTH = 100; +const ROSTER_LIMIT_COLLAPSED = 4; + +function stateToIcon(state: VibeSessionState): ToolUIStatus { + switch (state) { + case "running": + return "running"; + case "starting": + return "pending"; + case "idle": + return "done"; + case "dead": + return "aborted"; + } +} + +function stateToColor(state: VibeSessionState): ToolUIColor { + switch (state) { + case "running": + return "accent"; + case "starting": + return "accent"; + case "idle": + return "success"; + case "dead": + return "muted"; + } +} + +function rosterLine(entry: VibeRosterEntry, uiTheme: Theme, spinnerFrame: number | undefined): string { + const icon = formatStatusIcon( + stateToIcon(entry.state), + uiTheme, + entry.state === "running" ? spinnerFrame : undefined, + ); + const badge = formatBadge(entry.cli, stateToColor(entry.state), uiTheme); + const gist = entry.lastActivity ? ` ${uiTheme.fg("dim", oneLineLabel(replaceTabs(entry.lastActivity), 60))}` : ""; + const turns = uiTheme.fg("muted", `${entry.turns}t${entry.queued > 0 ? `+${entry.queued}q` : ""}`); + return truncateToWidth( + `${icon} ${badge} ${uiTheme.fg("toolOutput", entry.id)} ${uiTheme.fg("dim", entry.state)} ${turns}${gist}`, + ROSTER_LINE_WIDTH, + Ellipsis.Unicode, + ); +} + +interface VibeRenderArgs { + cli?: VibeCli; + prompt?: string; + session?: string; + message?: string; + sessions?: string[]; +} + +function describeCall(op: VibeOp, args: VibeRenderArgs | undefined): string { + switch (op) { + case "spawn": + return `spawn ${args?.cli ?? "?"}${args?.prompt ? `: ${oneLineLabel(args.prompt, 60)}` : ""}`; + case "send": + return `send → ${args?.session ?? "?"}${args?.message ? `: ${oneLineLabel(args.message, 60)}` : ""}`; + case "wait": + return args?.sessions?.length ? `wait on ${args.sessions.join(", ")}` : "wait on running sessions"; + case "kill": + return `kill ${args?.session ?? "?"}`; + case "list": + return "sessions"; + } +} + +/** Build the shared vibe renderer for one tool name. */ +export function createVibeToolRenderer(op: VibeOp) { + return { + inline: true, + mergeCallAndResult: true, + + renderCall(args: VibeRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component { + return new Text(renderStatusLine({ icon: "pending", title: `vibe ${describeCall(op, args)}` }, uiTheme), 0, 0); + }, + + renderResult( + result: { content: Array<{ type: string; text?: string }>; details?: VibeToolDetails; isError?: boolean }, + options: RenderResultOptions, + uiTheme: Theme, + args?: VibeRenderArgs, + ): Component { + const details = result.details; + const title = `vibe ${describeCall(op, args)}`; + if (!details || result.isError) { + const fallback = result.content.find(part => part.type === "text")?.text ?? ""; + const header = renderStatusLine({ icon: result.isError ? "error" : "done", title }, uiTheme); + const body = fallback + ? `\n ${uiTheme.fg(result.isError ? "error" : "dim", oneLineLabel(replaceTabs(fallback), ROSTER_LINE_WIDTH))}` + : ""; + return new Text(`${header}${body}`, 0, 0); + } + + const running = details.roster.filter(entry => entry.state === "running" || entry.state === "starting").length; + const meta: string[] = []; + if (running > 0) meta.push(uiTheme.fg("accent", `${running} running`)); + if (details.wait?.timedOut) meta.push(uiTheme.fg("warning", "timed out")); + if (details.wait?.settled.length) meta.push(uiTheme.fg("success", `${details.wait.settled.length} settled`)); + const header = renderStatusLine( + { + icon: details.wait?.timedOut ? "warning" : "done", + spinnerFrame: running > 0 ? options.spinnerFrame : undefined, + title, + meta, + }, + uiTheme, + ); + + const roster = options.expanded ? details.roster : details.roster.slice(0, ROSTER_LIMIT_COLLAPSED); + const lines = roster.map(entry => ` ${rosterLine(entry, uiTheme, options.spinnerFrame)}`); + if (!options.expanded && details.roster.length > roster.length) { + lines.push(` ${uiTheme.fg("dim", `… ${details.roster.length - roster.length} more`)}`); + } + return new Text([header, ...lines].join("\n"), 0, 0); + }, + }; +} diff --git a/packages/coding-agent/src/vibe/runtime.ts b/packages/coding-agent/src/vibe/runtime.ts new file mode 100644 index 000000000..5eeddaf19 --- /dev/null +++ b/packages/coding-agent/src/vibe/runtime.ts @@ -0,0 +1,635 @@ +/** + * Vibe mode worker-session runtime. + * + * Owns the persistent, addressable worker sessions ("CLIs") the vibe director + * drives. Each worker is a real task-executor subagent with full tool access: + * spawned once through {@link runSubprocess} (keep-alive), continued + * turn-by-turn through {@link runSubagentFollowUpTurn}. Between turns the + * worker lives in the AgentRegistry / AgentLifecycleManager as an adopted idle + * agent (TTL park + JSONL revive), so its conversation context survives across + * turns and even across parking. + * + * Every turn runs as an AsyncJobManager job, so a completed turn self-delivers + * into the director's conversation exactly like an async `task` result, and + * `vibe_wait` can block on the first settling turn with `job`-poll semantics. + */ +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { logger, prompt, Snowflake } from "@oh-my-pi/pi-utils"; +import type { AsyncJob, AsyncJobManager } from "../async/job-manager"; +import { resolveAgentModelPatterns } from "../config/model-resolver"; +import type { LocalProtocolOptions } from "../internal-urls"; +import { registerArtifactsDir } from "../internal-urls/registry-helpers"; +import { MCPManager } from "../mcp/manager"; +import vibeTurnResultTemplate from "../prompts/tools/vibe-turn-result.md" with { type: "text" }; +import { AgentLifecycleManager } from "../registry/agent-lifecycle"; +import { AgentRegistry, MAIN_AGENT_ID } from "../registry/agent-registry"; +import { getBundledAgent } from "../task/agents"; +import { type ExecutorOptions, runSubagentFollowUpTurn, runSubprocess } from "../task/executor"; +import { generateTaskName } from "../task/name-generator"; +import { AgentOutputManager } from "../task/output-manager"; +import { type AgentDefinition, type AgentProgress, oneLineLabel, type SingleResult } from "../task/types"; +import type { ToolSession } from "../tools"; +import { formatDuration } from "../tools/render-utils"; +import { ToolError } from "../tools/tool-errors"; + +/** The two worker CLI flavors the director drives. */ +export type VibeCli = "fast" | "good"; + +/** + * CLI flavor → bundled agent type. This IS the model-tier mapping: `sonic` + * carries `model: "pi/smol"` (the configured fast/low-latency role) and `task` + * carries `model: "pi/task"` (inherits the session's strong model). + * Resolution goes through {@link resolveAgentModelPatterns} exactly like a + * `task` spawn, so `task.agentModelOverrides` and model-role settings apply. + */ +export const VIBE_CLI_AGENT: Record = { + fast: "sonic", + good: "task", +}; + +/** Worker session lifecycle as shown to the director. */ +export type VibeSessionState = "starting" | "running" | "idle" | "dead"; + +/** One completed tool call in the per-turn activity trace. */ +interface VibeTraceEntry { + tool: string; + args: string; + endMs: number; +} + +/** Cap on trace entries retained per turn (the run monitor keeps 5; we widen the window). */ +const TURN_TRACE_CAP = 40; +/** Cap on a single rendered trace line. */ +const TRACE_LINE_MAX = 120; +/** Default `vibe_wait` window when no timeout was given (ms). */ +const DEFAULT_WAIT_TIMEOUT_MS = 30_000; +/** Response text cap inside a delivered turn result; full output stays at agent://. */ +const RESPONSE_PREVIEW_MAX = 6000; + +interface VibeTurn { + jobId: string; + message: string; + startedAt: number; + /** Trace of tool calls completed during this turn, oldest first. */ + trace: VibeTraceEntry[]; + /** Total completed tool calls (trace may be narrower than this). */ + toolCount: number; +} + +interface VibeRecord { + id: string; + cli: VibeCli; + ownerId: string; + agent: AgentDefinition; + modelOverride?: string | string[]; + state: VibeSessionState; + createdAt: number; + lastActivityAt: number; + /** One-line gist of the latest activity (intent, tool, or result preview). */ + lastActivity?: string; + /** Resolved model display string once known. */ + resolvedModel?: string; + turn?: VibeTurn; + /** Job id of the most recently settled turn (wait snapshots after settle). */ + lastJobId?: string; + /** Messages queued while a turn was in flight; drained into the next turn. */ + queue: string[]; + turnCount: number; + killed: boolean; +} + +/** Roster entry returned by {@link VibeSessionRegistry.list}. */ +export interface VibeRosterEntry { + id: string; + cli: VibeCli; + state: VibeSessionState; + model?: string; + turns: number; + queued: number; + lastActivity?: string; + lastActivityAt: number; + /** Job id of the in-flight turn, when running. */ + jobId?: string; +} + +export interface VibeSpawnOutcome { + id: string; + jobId: string; +} + +export interface VibeSendOutcome { + id: string; + /** + * - `turn`: a new background turn was started (`jobId` set). + * - `steered`: worker was mid-turn and streaming; delivered as steering. + * - `queued`: worker was mid-turn but not steerable; drained into the next turn. + */ + mode: "turn" | "steered" | "queued"; + jobId?: string; +} + +export interface VibeKillOutcome { + id: string; + /** True when an in-flight turn job was cancelled along the way. */ + cancelledTurn: boolean; +} + +export interface VibeWaitOutcome { + /** Watched sessions whose latest turn settled during (or before) the wait. */ + settled: Array<{ id: string; jobId: string; status: "completed" | "failed" | "cancelled"; resultText: string }>; + /** Watched sessions still mid-turn when the wait returned. */ + stillRunning: string[]; + timedOut: boolean; +} + +/** Normalize a text fragment to one bounded roster/trace line. */ +function firstLine(text: string, max = 100): string { + return oneLineLabel(text, max); +} + +/** Merge the monitor's rolling `recentTools` window (newest first) into the per-turn trace (oldest first). */ +function mergeTrace(turn: VibeTurn, progress: AgentProgress): void { + turn.toolCount = progress.toolCount; + for (let i = progress.recentTools.length - 1; i >= 0; i--) { + const entry = progress.recentTools[i]; + if (turn.trace.some(seen => seen.endMs === entry.endMs && seen.tool === entry.tool && seen.args === entry.args)) { + continue; + } + turn.trace.push({ tool: entry.tool, args: entry.args, endMs: entry.endMs }); + if (turn.trace.length > TURN_TRACE_CAP) turn.trace.shift(); + } +} + +/** Thrown from a turn job body so the job manager marks the job failed while carrying the formatted result. */ +export class VibeTurnError extends Error {} + +/** + * Process-global registry of vibe worker sessions, scoped per owner agent id + * (same convention as AsyncJobManager owner filters). The interactive mode + * kills an owner's sessions on vibe-mode exit via {@link killAll}. + */ +export class VibeSessionRegistry { + static #global: VibeSessionRegistry | undefined; + + static global(): VibeSessionRegistry { + if (!VibeSessionRegistry.#global) { + VibeSessionRegistry.#global = new VibeSessionRegistry(); + } + return VibeSessionRegistry.#global; + } + + /** Reset the global registry. Test-only. */ + static resetGlobalForTests(): void { + VibeSessionRegistry.#global = undefined; + } + + readonly #records = new Map(); + + #manager(session: ToolSession): AsyncJobManager { + const manager = session.asyncJobManager; + if (!manager) { + throw new ToolError("Vibe sessions require async execution (no background job manager is available)."); + } + return manager; + } + + #record(owner: string, id: string): VibeRecord { + const record = this.#records.get(id.trim()); + if (!record || record.ownerId !== owner) { + const roster = this.listIds(owner); + throw new ToolError( + `Unknown vibe session "${id}".${roster.length > 0 ? ` Active sessions: ${roster.join(", ")}` : " No sessions — spawn one with vibe_spawn."}`, + ); + } + return record; + } + + listIds(owner: string): string[] { + const ids: string[] = []; + for (const record of this.#records.values()) { + if (record.ownerId === owner && record.state !== "dead") ids.push(record.id); + } + return ids; + } + + list(owner: string): VibeRosterEntry[] { + const entries: VibeRosterEntry[] = []; + for (const record of this.#records.values()) { + if (record.ownerId !== owner) continue; + entries.push({ + id: record.id, + cli: record.cli, + state: record.state, + model: record.resolvedModel, + turns: record.turnCount, + queued: record.queue.length, + lastActivity: record.lastActivity, + lastActivityAt: record.lastActivityAt, + jobId: record.turn?.jobId, + }); + } + return entries.sort((a, b) => a.lastActivityAt - b.lastActivityAt); + } + + /** Spawn a persistent worker session and start its first turn in the background. */ + async spawn(session: ToolSession, args: { cli: VibeCli; name?: string; prompt: string }): Promise { + const owner = session.getAgentId?.() ?? MAIN_AGENT_ID; + const manager = this.#manager(session); + const agentName = VIBE_CLI_AGENT[args.cli]; + const agent = getBundledAgent(agentName); + if (!agent) { + throw new ToolError(`Bundled agent "${agentName}" for vibe cli "${args.cli}" is unavailable.`); + } + + const agentModelOverrides = session.settings.get("task.agentModelOverrides"); + const modelOverride = resolveAgentModelPatterns({ + settingsOverride: agentModelOverrides[agentName], + agentModel: agent.model, + settings: session.settings, + activeModelPattern: session.getActiveModelString?.(), + fallbackModelPattern: session.getModelString?.(), + }); + + if (!session.agentOutputManager) { + session.agentOutputManager = new AgentOutputManager(session.getArtifactsDir ?? (() => null)); + } + const requestedName = args.name?.replace(/[^A-Za-z0-9_-]+/g, "").slice(0, 48); + const id = await session.agentOutputManager.allocate(requestedName || generateTaskName()); + + const record: VibeRecord = { + id, + cli: args.cli, + ownerId: owner, + agent, + modelOverride, + state: "starting", + createdAt: Date.now(), + lastActivityAt: Date.now(), + queue: [], + turnCount: 0, + killed: false, + }; + this.#records.set(id, record); + + try { + const jobId = this.#registerTurnJob(session, manager, record, args.prompt, { first: true }); + return { id, jobId }; + } catch (error) { + this.#records.delete(id); + throw error; + } + } + + /** + * Send a message to a worker. Mid-turn and streaming → steering; mid-turn + * otherwise → queued for the next turn; idle/parked → starts a new + * background turn immediately. + */ + async send(session: ToolSession, args: { session: string; message: string }): Promise { + const owner = session.getAgentId?.() ?? MAIN_AGENT_ID; + const record = this.#record(owner, args.session); + if (record.state === "dead") { + throw new ToolError(`Vibe session "${record.id}" is dead. Spawn a new one with vibe_spawn.`); + } + const message = args.message.trim(); + if (!message) throw new ToolError("Message must not be empty."); + + if (record.turn) { + const live = AgentRegistry.global().get(record.id)?.session; + if (live?.isStreaming) { + await live.steer(message); + record.lastActivityAt = Date.now(); + return { id: record.id, mode: "steered" }; + } + record.queue.push(message); + record.lastActivityAt = Date.now(); + return { id: record.id, mode: "queued" }; + } + + const manager = this.#manager(session); + const jobId = this.#registerTurnJob(session, manager, record, message, { first: false }); + return { id: record.id, mode: "turn", jobId }; + } + + /** + * Block until one watched session's in-flight turn settles, the timeout + * elapses, or `signal` aborts — `job` poll semantics. Settled turns are + * acknowledged against the job manager so their results are not delivered + * a second time as async follow-ups. + */ + async wait( + session: ToolSession, + args: { sessions?: string[]; timeoutMs?: number; signal?: AbortSignal }, + ): Promise { + const owner = session.getAgentId?.() ?? MAIN_AGENT_ID; + const manager = this.#manager(session); + // Named sessions are watched regardless of state (a just-settled turn is + // reported from its retained job); the no-args form watches every + // session with a turn actually in flight. + const watched = args.sessions?.length + ? args.sessions.map(id => this.#record(owner, id)) + : [...this.#records.values()].filter(record => record.ownerId === owner && record.turn !== undefined); + + const collectSettled = (): VibeWaitOutcome["settled"] => { + const settled: VibeWaitOutcome["settled"] = []; + for (const record of watched) { + const jobId = record.turn?.jobId ?? record.lastJobId; + if (!jobId) continue; + const job = manager.getJob(jobId); + if (!job || job.status === "running") continue; + settled.push({ + id: record.id, + jobId, + status: job.status, + resultText: job.resultText ?? job.errorText ?? "(no output)", + }); + } + return settled; + }; + + const runningJobs: AsyncJob[] = []; + for (const record of watched) { + if (!record.turn) continue; + const job = manager.getJob(record.turn.jobId); + if (job?.status === "running") runningJobs.push(job); + } + + let waited = false; + if (runningJobs.length > 0 && collectSettled().length === 0) { + waited = true; + const timeoutMs = Math.max(1, Math.trunc(args.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS)); + const watchedJobIds = runningJobs.map(job => job.id); + manager.watchJobs(watchedJobIds); + const { promise: timeoutPromise, resolve: timeoutResolve } = Promise.withResolvers(); + const timeoutHandle = setTimeout(() => timeoutResolve(), timeoutMs); + const racePromises: Promise[] = [...runningJobs.map(job => job.promise), timeoutPromise]; + let abortCleanup: (() => void) | undefined; + if (args.signal) { + const { promise: abortPromise, resolve: abortResolve } = Promise.withResolvers(); + const onAbort = () => abortResolve(); + args.signal.addEventListener("abort", onAbort, { once: true }); + abortCleanup = () => args.signal?.removeEventListener("abort", onAbort); + racePromises.push(abortPromise); + } + try { + await Promise.race(racePromises); + } finally { + manager.unwatchJobs(watchedJobIds); + clearTimeout(timeoutHandle); + abortCleanup?.(); + } + } + + const settled = collectSettled(); + manager.acknowledgeDeliveries(settled.map(entry => entry.jobId)); + const settledIds = new Set(settled.map(entry => entry.id)); + const stillRunning = watched + .filter(record => !settledIds.has(record.id) && record.turn !== undefined) + .map(record => record.id); + return { settled, stillRunning, timedOut: waited && settled.length === 0 }; + } + + /** Terminate a worker: cancel its in-flight turn and dispose + unregister its session. */ + async kill(session: ToolSession, id: string): Promise { + const owner = session.getAgentId?.() ?? MAIN_AGENT_ID; + const record = this.#record(owner, id); + return this.#killRecord(record, session.asyncJobManager); + } + + /** Kill every session belonging to `owner` (vibe-mode exit / teardown). Returns the number killed. */ + async killAll(owner: string, manager?: AsyncJobManager): Promise { + let killed = 0; + for (const record of this.#records.values()) { + if (record.ownerId !== owner || record.state === "dead") continue; + await this.#killRecord(record, manager); + killed++; + } + return killed; + } + + async #killRecord(record: VibeRecord, manager: AsyncJobManager | undefined): Promise { + record.killed = true; + record.queue.length = 0; + let cancelledTurn = false; + if (record.turn && manager) { + cancelledTurn = manager.cancel(record.turn.jobId, { ownerId: record.ownerId }); + } + record.state = "dead"; + record.lastActivityAt = Date.now(); + record.lastActivity = "killed"; + try { + await AgentLifecycleManager.global().release(record.id); + } catch (error) { + logger.warn("vibe: failed to release worker session", { + id: record.id, + error: error instanceof Error ? error.message : String(error), + }); + } + return { id: record.id, cancelledTurn }; + } + + /** Build the ExecutorOptions for a first spawn, mirroring the `task`/eval-bridge plumbing. */ + async #buildSpawnOptions( + session: ToolSession, + record: VibeRecord, + message: string, + signal: AbortSignal, + onProgress: (progress: AgentProgress) => void, + ): Promise { + const sessionFile = session.getSessionFile(); + const sessionArtifactsDir = sessionFile ? sessionFile.slice(0, -6) : null; + const artifactsDir = sessionArtifactsDir ?? path.join(os.tmpdir(), `omp-vibe-${Snowflake.next()}`); + await fs.mkdir(artifactsDir, { recursive: true }); + if (!sessionArtifactsDir) registerArtifactsDir(artifactsDir); + const localProtocolOptions: LocalProtocolOptions = session.localProtocolOptions ?? { + getArtifactsDir: session.getArtifactsDir ?? (() => null), + getSessionId: session.getSessionId ?? (() => null), + }; + return { + cwd: session.cwd, + agent: record.agent, + task: message, + assignment: message, + description: `vibe ${record.cli} session`, + index: 0, + id: record.id, + taskDepth: session.taskDepth ?? 0, + detached: true, + modelOverride: record.modelOverride, + parentActiveModelPattern: session.getActiveModelString?.(), + thinkingLevel: record.agent.thinkingLevel, + sessionFile, + persistArtifacts: Boolean(sessionFile), + artifactsDir, + enableLsp: (session.enableLsp ?? true) && session.settings.get("task.enableLsp"), + signal, + eventBus: session.eventBus, + onProgress, + authStorage: session.authStorage, + modelRegistry: session.modelRegistry, + settings: session.settings, + mcpManager: session.mcpManager ?? MCPManager.instance(), + contextFiles: session.contextFiles?.filter(file => path.basename(file.path).toLowerCase() !== "agents.md"), + skills: [...(session.skills ?? [])], + workspaceTree: session.workspaceTree, + promptTemplates: session.promptTemplates, + rules: session.rules, + preloadedExtensionPaths: session.extensionPaths, + preloadedCustomToolPaths: session.customToolPaths, + localProtocolOptions, + parentArtifactManager: session.getArtifactManager?.() ?? undefined, + parentHindsightSessionState: session.getHindsightSessionState?.(), + parentMnemopiSessionState: session.getMnemopiSessionState?.(), + parentTelemetry: session.getTelemetry?.(), + parentEvalSessionId: session.getEvalSessionId?.() ?? undefined, + parentAgentId: session.getAgentId?.() ?? MAIN_AGENT_ID, + parentServiceTier: session.getServiceTierByFamily ? (session.getServiceTierByFamily() ?? null) : undefined, + keepAlive: true, + }; + } + + /** Register one background job that runs a single worker turn and self-delivers its result. */ + #registerTurnJob( + session: ToolSession, + manager: AsyncJobManager, + record: VibeRecord, + message: string, + options: { first: boolean }, + ): string { + const turnIndex = record.turnCount + 1; + const turn: VibeTurn = { + jobId: "", + message, + startedAt: Date.now(), + trace: [], + toolCount: 0, + }; + const onProgress = (progress: AgentProgress): void => { + mergeTrace(turn, progress); + record.resolvedModel = progress.resolvedModel ?? record.resolvedModel; + const gist = + progress.lastIntent ?? + (progress.currentTool ? `${progress.currentTool} ${progress.currentToolArgs ?? ""}` : undefined); + if (gist) record.lastActivity = firstLine(gist); + record.lastActivityAt = Date.now(); + }; + + const jobId = manager.register( + "task", + `vibe ${record.cli} ${record.id}: ${firstLine(message, 60)}`, + async ({ jobId: ownJobId, signal }) => { + record.state = "running"; + record.turnCount = turnIndex; + record.lastActivityAt = Date.now(); + try { + const result = options.first + ? await runSubprocess(await this.#buildSpawnOptions(session, record, message, signal, onProgress)) + : await runSubagentFollowUpTurn({ + id: record.id, + agent: record.agent, + message, + description: `vibe ${record.cli} session`, + signal, + onProgress, + eventBus: session.eventBus, + artifactsDir: session.getSessionFile()?.slice(0, -6), + }); + return this.#settleTurn(session, manager, record, turn, ownJobId, turnIndex, result); + } catch (error) { + if (error instanceof VibeTurnError) throw error; + this.#finishTurn(session, manager, record, ownJobId); + const reason = error instanceof Error ? error.message : String(error); + record.lastActivity = firstLine(`turn failed: ${reason}`); + throw new VibeTurnError( + `[vibe:${record.id} cli=${record.cli} turn=${turnIndex}] turn failed: ${reason}`, + ); + } + }, + { id: `${record.id}-t${turnIndex}`, ownerId: record.ownerId }, + ); + turn.jobId = jobId; + record.turn = turn; + return jobId; + } + + /** Post-turn bookkeeping shared by success and failure paths: clear the in-flight turn, flush the queue. */ + #finishTurn(session: ToolSession, manager: AsyncJobManager, record: VibeRecord, settledJobId: string): void { + record.lastJobId = settledJobId; + record.turn = undefined; + record.lastActivityAt = Date.now(); + if (record.killed) { + record.state = "dead"; + return; + } + // A spawn that failed before its session ever registered leaves nothing + // to continue — mark the record dead so sends fail with clear guidance. + record.state = AgentRegistry.global().get(record.id) ? "idle" : "dead"; + if (record.state === "dead" || record.queue.length === 0) return; + const nextMessage = record.queue.splice(0, record.queue.length).join("\n\n"); + try { + this.#registerTurnJob(session, manager, record, nextMessage, { first: false }); + } catch (error) { + // Leave the messages recoverable: a later vibe_send flushes again. + record.queue.unshift(nextMessage); + logger.warn("vibe: failed to start queued follow-up turn", { + id: record.id, + error: error instanceof Error ? error.message : String(error), + }); + } + } + + /** Format a settled turn into the self-delivering result text (activity trace + response). */ + #settleTurn( + session: ToolSession, + manager: AsyncJobManager, + record: VibeRecord, + turn: VibeTurn, + settledJobId: string, + turnIndex: number, + result: SingleResult, + ): string { + this.#finishTurn(session, manager, record, settledJobId); + const failed = result.exitCode !== 0 || result.aborted === true; + const status = result.aborted ? "aborted" : failed ? "failed" : "completed"; + record.lastActivity = firstLine( + failed + ? `turn ${turnIndex} ${status}: ${result.abortReason ?? result.error ?? ""}` + : (result.lastIntent ?? result.output), + ); + + const traceLines = turn.trace.map(entry => + firstLine(`${entry.tool}${entry.args ? `(${entry.args})` : ""}`, TRACE_LINE_MAX), + ); + const traceOverflow = Math.max(0, turn.toolCount - turn.trace.length); + let response = result.output.trim() || "(no output)"; + let responseTruncated = false; + if (response.length > RESPONSE_PREVIEW_MAX) { + const slice = response.slice(0, RESPONSE_PREVIEW_MAX); + const lastNewline = slice.lastIndexOf("\n"); + response = lastNewline > 0 ? slice.slice(0, lastNewline) : slice; + responseTruncated = true; + } + const text = prompt + .render(vibeTurnResultTemplate, { + id: record.id, + cli: record.cli, + turn: turnIndex, + status, + duration: formatDuration(result.durationMs), + requests: result.requests, + toolCount: turn.toolCount, + model: result.resolvedModel ?? record.resolvedModel ?? "", + trace: traceLines, + traceOverflow: traceOverflow > 0 ? traceOverflow : undefined, + response, + responseTruncated, + error: failed ? (result.abortReason ?? result.error ?? result.stderr ?? "") : "", + alive: record.state !== "dead", + }) + .trim(); + if (failed) throw new VibeTurnError(text); + return text; + } +} diff --git a/packages/coding-agent/src/vibe/state.ts b/packages/coding-agent/src/vibe/state.ts new file mode 100644 index 000000000..a896b97d0 --- /dev/null +++ b/packages/coding-agent/src/vibe/state.ts @@ -0,0 +1,4 @@ +/** Vibe mode session-level state, mirroring {@link ../plan-mode/state.ts}. */ +export interface VibeModeState { + enabled: boolean; +} From 1ab9c367ede51f9eebff31e9b3e0f6ddbc0ad97b Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 02:12:08 +0200 Subject: [PATCH 128/205] feat(vibe): integrated vibe mode with interactive interface - Implemented Vibe mode to enable worker session management and director-role context injection. - Added a `/vibe` slash command and integrated status line UI to display mode activity. - Configured restricted toolsets and guards to prevent concurrent conflicts with existing Goal or Plan modes. - Provided system prompts and tool templates to support specialized agent communication and task orchestration. --- .../modes/components/status-line/component.ts | 6 + .../modes/components/status-line/segments.ts | 6 + .../src/modes/components/status-line/types.ts | 3 + .../src/modes/interactive-mode.ts | 123 +++++++++++++++++- packages/coding-agent/src/modes/types.ts | 2 + .../src/prompts/system/vibe-mode-active.md | 23 ++++ .../src/prompts/tools/vibe-kill.md | 3 + .../src/prompts/tools/vibe-list.md | 3 + .../src/prompts/tools/vibe-send.md | 9 ++ .../src/prompts/tools/vibe-spawn.md | 10 ++ .../src/prompts/tools/vibe-turn-result.md | 19 +++ .../src/prompts/tools/vibe-wait.md | 8 ++ packages/coding-agent/src/sdk.ts | 14 ++ .../coding-agent/src/session/agent-session.ts | 42 ++++++ .../src/slash-commands/builtin-registry.ts | 16 +++ packages/coding-agent/src/tools/index.ts | 7 + packages/coding-agent/src/tools/renderers.ts | 6 + 17 files changed, 299 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/src/prompts/system/vibe-mode-active.md create mode 100644 packages/coding-agent/src/prompts/tools/vibe-kill.md create mode 100644 packages/coding-agent/src/prompts/tools/vibe-list.md create mode 100644 packages/coding-agent/src/prompts/tools/vibe-send.md create mode 100644 packages/coding-agent/src/prompts/tools/vibe-spawn.md create mode 100644 packages/coding-agent/src/prompts/tools/vibe-turn-result.md create mode 100644 packages/coding-agent/src/prompts/tools/vibe-wait.md diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts index 195539a46..e709f9d78 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.ts @@ -275,6 +275,7 @@ export class StatusLineComponent implements Component { #planModeStatus: { enabled: boolean; paused: boolean } | null = null; #loopModeStatus: { enabled: boolean } | null = null; #goalModeStatus: { enabled: boolean; paused: boolean } | null = null; + #vibeModeStatus: { enabled: boolean } | null = null; #collabStatus: CollabStatus | null = null; #focusedAgentId: string | undefined; #activeRepoCache: ActiveRepoCache | undefined; @@ -503,6 +504,10 @@ export class StatusLineComponent implements Component { this.#goalModeStatus = status ?? null; } + setVibeModeStatus(status: { enabled: boolean } | undefined): void { + this.#vibeModeStatus = status ?? null; + } + setCollabStatus(status: CollabStatus | null): void { this.#collabStatus = status; } @@ -1047,6 +1052,7 @@ export class StatusLineComponent implements Component { planMode: this.#planModeStatus, loopMode: this.#loopModeStatus, goalMode: this.#goalModeStatus, + vibeMode: this.#vibeModeStatus, collab: this.#collabStatus, usageStats, contextPercent, diff --git a/packages/coding-agent/src/modes/components/status-line/segments.ts b/packages/coding-agent/src/modes/components/status-line/segments.ts index 95a36814a..e1ef2b87e 100644 --- a/packages/coding-agent/src/modes/components/status-line/segments.ts +++ b/packages/coding-agent/src/modes/components/status-line/segments.ts @@ -213,6 +213,12 @@ const modeSegment: StatusLineSegment = { return renderGoalMode(ctx, goal); } + const vibe = ctx.vibeMode; + if (vibe?.enabled) { + const content = withIcon(theme.icon.agents, "Vibe"); + return { content: theme.fg("accent", content), visible: true }; + } + const loop = ctx.loopMode; if (loop?.enabled) { const content = withIcon(theme.icon.loop, "Loop"); diff --git a/packages/coding-agent/src/modes/components/status-line/types.ts b/packages/coding-agent/src/modes/components/status-line/types.ts index ae02e7168..8fe16380d 100644 --- a/packages/coding-agent/src/modes/components/status-line/types.ts +++ b/packages/coding-agent/src/modes/components/status-line/types.ts @@ -67,6 +67,9 @@ export interface SegmentContext { enabled: boolean; paused: boolean; } | null; + vibeMode: { + enabled: boolean; + } | null; collab: CollabStatus | null; // Cached values for performance (computed once per render) usageStats: { diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 5bd64e097..bbdad89b1 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -92,7 +92,7 @@ import planModeApprovedPrompt from "../prompts/system/plan-mode-approved.md" wit import planModeCompactInstructionsPrompt from "../prompts/system/plan-mode-compact-instructions.md" with { type: "text", }; -import type { AgentRegistry } from "../registry/agent-registry"; +import { type AgentRegistry, MAIN_AGENT_ID } from "../registry/agent-registry"; import { type AgentSession, type AgentSessionEvent, @@ -118,6 +118,7 @@ import { setAutoQaConsentHandler } from "../tools/report-tool-issue"; import { type ResolveToolDetails, runResolveInvocation } from "../tools/resolve"; import { formatPhaseDisplayName, todoMatchesAnyDescription } from "../tools/todo"; import { ToolError } from "../tools/tool-errors"; +import { VIBE_TOOL_NAMES } from "../tools/vibe"; import { vocalizer } from "../tts/vocalizer"; import { renderTreeList } from "../tui/tree-list"; import type { EventBus } from "../utils/event-bus"; @@ -125,6 +126,7 @@ import { getEditorCommand, openInEditor } from "../utils/external-editor"; import { getSessionAccentAnsi, getSessionAccentHex } from "../utils/session-color"; import { messageHasDisplayableThinking } from "../utils/thinking-display"; import { popTerminalTitle, pushTerminalTitle, setSessionTerminalTitle } from "../utils/title-generator"; +import { VibeSessionRegistry } from "../vibe/runtime"; import type { AssistantMessageComponent } from "./components/assistant-message"; import type { BashExecutionComponent } from "./components/bash-execution"; import { ChatBlock, type ChatBlockHost } from "./components/chat-block"; @@ -434,6 +436,7 @@ export class InteractiveMode implements InteractiveModeContext { planModePaused = false; goalModeEnabled = false; goalModePaused = false; + vibeModeEnabled = false; planModePlanFilePath: string | undefined = undefined; loopModeEnabled = false; loopPrompt: string | undefined = undefined; @@ -528,6 +531,7 @@ export class InteractiveMode implements InteractiveModeContext { readonly #changelogMarkdown: string | undefined; #planModePreviousTools: string[] | undefined; #goalModePreviousTools: string[] | undefined; + #vibeModePreviousTools: string[] | undefined; #goalContinuationTimer: NodeJS.Timeout | undefined; #goalTurnHadToolCalls = false; #goalContinuationTurnInFlight = false; @@ -1943,6 +1947,11 @@ export class InteractiveMode implements InteractiveModeContext { this.ui.requestRender(); } + #updateVibeModeStatus(): void { + this.statusLine.setVibeModeStatus(this.vibeModeEnabled ? { enabled: true } : undefined); + this.ui.requestRender(); + } + #updateGoalModeStatus(): void { const status = this.goalModeEnabled || this.goalModePaused @@ -2111,6 +2120,20 @@ export class InteractiveMode implements InteractiveModeContext { this.#cancelGoalContinuation(); this.#updateGoalModeStatus(); } + + if (this.vibeModeEnabled) { + if (this.#vibeModePreviousTools !== undefined) { + await this.session.setActiveToolsByName(this.#vibeModePreviousTools); + } + this.session.setVibeModeState(undefined); + this.vibeModeEnabled = false; + this.#vibeModePreviousTools = undefined; + await VibeSessionRegistry.global().killAll( + this.session.getAgentId() ?? MAIN_AGENT_ID, + this.session.asyncJobManager, + ); + this.#updateVibeModeStatus(); + } } /** Reconcile mode state from session entries on resume/switch. */ @@ -2150,6 +2173,10 @@ export class InteractiveMode implements InteractiveModeContext { return; } this.session.goalRuntime.clearAccounting(); + if (sessionContext.mode === "vibe") { + await this.#enterVibeMode(); + return; + } if (!this.session.settings.get("plan.enabled")) { // Clear stale plan/plan_paused mode so re-enabling the setting // later doesn't unexpectedly restore an old plan session. @@ -2176,6 +2203,10 @@ export class InteractiveMode implements InteractiveModeContext { this.showWarning("Exit goal mode first."); return; } + if (this.vibeModeEnabled) { + this.showWarning("Exit vibe mode first."); + return; + } this.planModePaused = false; @@ -2343,6 +2374,10 @@ export class InteractiveMode implements InteractiveModeContext { this.showWarning("Exit plan mode first."); return; } + if (this.vibeModeEnabled) { + this.showWarning("Exit vibe mode first."); + return; + } const previousTools = this.session.getActiveToolNames().filter(name => name !== "goal"); const goalTools = [...new Set([...previousTools, "goal"])]; this.#goalModePreviousTools = previousTools; @@ -2847,6 +2882,10 @@ export class InteractiveMode implements InteractiveModeContext { this.showWarning("Exit goal mode first."); return; } + if (this.vibeModeEnabled) { + this.showWarning("Exit vibe mode first."); + return; + } if (this.planModeEnabled) { const planFilePath = this.planModePlanFilePath ?? (await this.#getPlanFilePath()); if (await this.#hasPlanModeDraftContent(planFilePath)) { @@ -2882,6 +2921,84 @@ export class InteractiveMode implements InteractiveModeContext { } } + /** + * `/vibe` toggle. Entering strips the active toolset down to `read` plus the + * vibe session tools and injects the director context; exiting restores the + * previous toolset and kills every worker session (boring and safe — workers + * do not outlive the mode that directs them). + */ + async handleVibeModeCommand(initialPrompt?: string): Promise { + if (this.vibeModeEnabled) { + await this.#exitVibeMode(); + return; + } + if (this.planModeEnabled || this.planModePaused) { + this.showWarning("Exit plan mode first."); + return; + } + if (this.goalModeEnabled || this.goalModePaused) { + this.showWarning("Exit goal mode first."); + return; + } + await this.#enterVibeMode(); + if (initialPrompt && this.onInputCallback) { + this.onInputCallback(this.startPendingSubmission({ text: initialPrompt })); + } + } + + async #enterVibeMode(): Promise { + if (this.vibeModeEnabled) { + return; + } + if (this.planModeEnabled || this.planModePaused) { + this.showWarning("Exit plan mode first."); + return; + } + if (this.goalModeEnabled || this.goalModePaused) { + this.showWarning("Exit goal mode first."); + return; + } + + this.#vibeModePreviousTools = this.session.getActiveToolNames(); + this.vibeModeEnabled = true; + // Suppress cache-miss marker on the next turn: vibe mode changes the + // injected context, which predictably invalidates the cache. + this.lastAssistantUsage = undefined; + await this.session.setActiveToolsByName(["read", ...VIBE_TOOL_NAMES]); + this.session.setVibeModeState({ enabled: true }); + if (this.session.isStreaming) { + await this.session.sendVibeModeContext({ deliverAs: "steer" }); + } + this.#updateVibeModeStatus(); + this.sessionManager.appendModeChange("vibe"); + this.showStatus("Vibe mode enabled. You direct fast/good worker sessions; toolset is read + vibe tools."); + } + + async #exitVibeMode(): Promise { + if (!this.vibeModeEnabled) { + return; + } + const previousTools = this.#vibeModePreviousTools; + if (previousTools && previousTools.length > 0) { + await this.session.setActiveToolsByName(previousTools); + } + this.session.setVibeModeState(undefined); + this.vibeModeEnabled = false; + this.#vibeModePreviousTools = undefined; + this.lastAssistantUsage = undefined; + const killed = await VibeSessionRegistry.global().killAll( + this.session.getAgentId() ?? MAIN_AGENT_ID, + this.session.asyncJobManager, + ); + this.#updateVibeModeStatus(); + this.sessionManager.appendModeChange("none"); + this.showStatus( + killed > 0 + ? `Vibe mode disabled. Killed ${killed} worker session${killed === 1 ? "" : "s"}.` + : "Vibe mode disabled.", + ); + } + async #handleGoalBudgetCommand(rawBudget: string): Promise { const state = this.session.getGoalModeState(); if (!this.goalModeEnabled || !state?.enabled) { @@ -2914,6 +3031,10 @@ export class InteractiveMode implements InteractiveModeContext { this.showWarning("Exit plan mode first."); return; } + if (this.vibeModeEnabled) { + this.showWarning("Exit vibe mode first."); + return; + } if (!this.session.settings.get("goal.enabled")) { this.showWarning("Goal mode is disabled. Enable it in settings (goal.enabled)."); return; diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 8c66eb8b7..a842c3503 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -153,6 +153,7 @@ export interface InteractiveModeContext { toolOutputExpanded: boolean; todoExpanded: boolean; planModeEnabled: boolean; + vibeModeEnabled: boolean; goalModeEnabled: boolean; goalModePaused: boolean; loopModeEnabled: boolean; @@ -394,6 +395,7 @@ export interface InteractiveModeContext { openExternalEditor(): void; registerExtensionShortcuts(): void; handlePlanModeCommand(initialPrompt?: string): Promise; + handleVibeModeCommand(initialPrompt?: string): Promise; handleGoalModeCommand(rest?: string): Promise; handleGuidedGoalCommand(rest?: string): Promise; handleLoopCommand(args?: string): Promise; diff --git a/packages/coding-agent/src/prompts/system/vibe-mode-active.md b/packages/coding-agent/src/prompts/system/vibe-mode-active.md new file mode 100644 index 000000000..6d09f7b0d --- /dev/null +++ b/packages/coding-agent/src/prompts/system/vibe-mode-active.md @@ -0,0 +1,23 @@ + +Vibe mode is ON. You are the DIRECTOR. You do not edit, run, grep, or build anything yourself — your hands are off the keyboard. You drive two kinds of worker CLIs, each a full coding agent with every normal tool, and you verify their work by reading files. + +Your entire toolset: `read`, `vibe_spawn`, `vibe_send`, `vibe_wait`, `vibe_kill`, `vibe_list`. + +# The two CLIs you drive + +- `fast` — low-latency model. Mechanical, well-specified work: renames, small fixes, boilerplate, data collection, running tests and reporting output. +- `good` — strong model. Hard work: design, tricky debugging, multi-file refactors, anything needing judgment. + +Sessions are persistent conversations, like terminals you keep open. A session remembers everything you told it and everything it did. Spawn once per workstream, then keep talking to the SAME session — never respawn for a follow-up on the same workstream. + +# How to direct + +1. Split the request into independent workstreams. One session per workstream; keep each session on its own workstream to build useful context. +2. `vibe_spawn` with a complete, self-contained brief: files, constraints, acceptance criteria. Workers start blank — they never see this conversation. +3. Sends and spawns return immediately; results arrive on their own when a worker finishes its turn. Keep directing other sessions meanwhile; call `vibe_wait` only when you cannot proceed without a result. +4. When a turn result arrives, judge it: `read` the touched files to verify claims before building on them. Follow up with `vibe_send` — corrections, next step, or a review request. +5. Route by difficulty: draft with `fast`, escalate to `good` when `fast` stalls or the problem needs judgment; have `good` design and `fast` execute the mechanical parts. +6. `vibe_kill` a session that is stuck or whose workstream is done; `vibe_list` when you lose track of the roster. + +Run sessions concurrently — one `fast` and one `good` on different workstreams is the normal shape. You stay responsible for the final outcome: verify with `read`, do not take a worker's word for it. + diff --git a/packages/coding-agent/src/prompts/tools/vibe-kill.md b/packages/coding-agent/src/prompts/tools/vibe-kill.md new file mode 100644 index 000000000..397883a83 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/vibe-kill.md @@ -0,0 +1,3 @@ +Terminates a worker session: aborts its in-flight turn (if any) and discards the session. Its conversation cannot be continued afterwards — the transcript stays readable at `history://`. + +Kill sessions that are stuck, looping, or whose workstream is complete. Freeing dead weight keeps the roster legible. diff --git a/packages/coding-agent/src/prompts/tools/vibe-list.md b/packages/coding-agent/src/prompts/tools/vibe-list.md new file mode 100644 index 000000000..cea2c1d34 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/vibe-list.md @@ -0,0 +1,3 @@ +Shows your worker-session roster: id, CLI flavor (`fast`/`good`), state (`starting`/`running`/`idle`/`dead`), model, turn count, queued messages, and a one-line gist of each session's latest activity. + +Use it to reorient: which sessions exist, who is busy, who is idle and ready for the next instruction. diff --git a/packages/coding-agent/src/prompts/tools/vibe-send.md b/packages/coding-agent/src/prompts/tools/vibe-send.md new file mode 100644 index 000000000..5865d70ee --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/vibe-send.md @@ -0,0 +1,9 @@ +Sends a message to one of your worker sessions (by id from `vibe_spawn` / `vibe_list`). The session keeps its full conversation history — refer to earlier work naturally ("now do the same for the other module"). + +Returns immediately with an ack telling you how the message landed: + +- `turn` — the worker was idle; a new turn started. Its result self-delivers when done. +- `steered` — the worker was mid-turn; your message was injected into the running turn as live steering. +- `queued` — the worker was mid-turn and not steerable right now; your message runs as the next turn automatically. + +Use it for follow-ups, corrections, scope changes, and review requests. Never re-explain prior context — the session already has it. diff --git a/packages/coding-agent/src/prompts/tools/vibe-spawn.md b/packages/coding-agent/src/prompts/tools/vibe-spawn.md new file mode 100644 index 000000000..bc588e045 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/vibe-spawn.md @@ -0,0 +1,10 @@ +Starts a persistent worker session — a full coding agent (edit, bash, grep, everything) that you drive by conversation. Pick the CLI flavor per task: + +- `fast`: low-latency model for mechanical, well-specified work (renames, boilerplate, running tests, data collection). +- `good`: strong model for hard work (design, debugging, multi-file changes, judgment calls). + +`prompt` is the session's first instruction. The worker starts with NO context beyond it — include files, constraints, and acceptance criteria. `name` (optional) labels the session; otherwise one is generated. + +Returns immediately with the session id; the turn's result (activity trace + the worker's response) is delivered to you automatically when the worker finishes. Do not wait unless you are blocked — keep directing other sessions. + +The session persists after the turn: it remembers the whole conversation. Continue it with `vibe_send`; never spawn a second session for a follow-up on the same workstream. diff --git a/packages/coding-agent/src/prompts/tools/vibe-turn-result.md b/packages/coding-agent/src/prompts/tools/vibe-turn-result.md new file mode 100644 index 000000000..33ea47cd8 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/vibe-turn-result.md @@ -0,0 +1,19 @@ + + +{{#each trace}} +- {{this}} +{{/each}} +{{#if traceOverflow}} +- … {{traceOverflow}} earlier tool call(s) not shown +{{/if}} + + +{{response}} + +{{#if error}} +{{error}} +{{/if}} +{{#if alive}} +Session `{{id}}` is idle and retains this conversation — continue it with vibe_send. Transcript: history://{{id}} +{{/if}} + diff --git a/packages/coding-agent/src/prompts/tools/vibe-wait.md b/packages/coding-agent/src/prompts/tools/vibe-wait.md new file mode 100644 index 000000000..2c71d8b32 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/vibe-wait.md @@ -0,0 +1,8 @@ +Blocks until ONE watched session finishes its current turn, the timeout elapses, or you are interrupted — not until all finish. Re-issue to keep waiting. + +Turn results normally deliver themselves; you NEVER need this to receive output. Use it only when you are completely blocked and cannot direct any other session. + +- `sessions` — ids to watch. Omit to watch every session with a turn in flight. +- `timeout` — seconds to wait (default 30). + +A finished turn's full result (activity trace + response) is returned here and will not be re-delivered separately. diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 4d3fd949a..8c0a35153 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -184,6 +184,7 @@ import { setPreferredSearchProvider, type Tool, type ToolSession, + VIBE_TOOL_NAMES, WebSearchTool, WriteTool, warmupLspServers, @@ -2239,6 +2240,19 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} builtInRegistryToolNames.add(goalTool.name); } } + // Vibe tools are hidden from every default/discovery set; they exist in the + // registry so entering vibe mode can activate them via setActiveToolsByName. + // Top-level interactive sessions only — subagents never direct vibe workers. + if ((options.taskDepth ?? 0) === 0 && !options.parentTaskPrefix) { + for (const name of VIBE_TOOL_NAMES) { + if (toolRegistry.has(name)) continue; + const vibeTool = await logger.time(`createTools:${name}:session`, HIDDEN_TOOLS[name], toolSession); + if (vibeTool) { + toolRegistry.set(vibeTool.name, wrapToolWithMetaNotice(vibeTool)); + builtInRegistryToolNames.add(vibeTool.name); + } + } + } for (const tool of wrappedExtensionTools) { toolRegistry.set(tool.name, tool); builtInRegistryToolNames.delete(tool.name); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 8ec7c69c2..e34161e54 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -277,6 +277,7 @@ import toolCallLoopRedirectTemplate from "../prompts/system/tool-call-loop-redir import ttsrInterruptTemplate from "../prompts/system/ttsr-interrupt.md" with { type: "text" }; import ttsrToolReminderTemplate from "../prompts/system/ttsr-tool-reminder.md" with { type: "text" }; import unexpectedStopRetryTemplate from "../prompts/system/unexpected-stop-retry.md" with { type: "text" }; +import vibeModeActivePrompt from "../prompts/system/vibe-mode-active.md" with { type: "text" }; import { AgentRegistry } from "../registry/agent-registry"; import { deobfuscateAssistantContent, @@ -330,6 +331,7 @@ import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallba import { formatLocalCalendarDate } from "../utils/local-date"; import { generateSessionTitle } from "../utils/title-generator"; import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice"; +import type { VibeModeState } from "../vibe/state"; import type { AuthStorage } from "./auth-storage"; import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge"; import { @@ -1608,6 +1610,7 @@ export class AgentSession { #advisorPrimaryTurnsCompleted = 0; #advisorInterruptImmuneTurnStart: number | undefined; #planModeState: PlanModeState | undefined; + #vibeModeState: VibeModeState | undefined; #goalModeState: GoalModeState | undefined; #goalRuntime: GoalRuntime; #advisorEnabled = false; @@ -7071,6 +7074,14 @@ export class AgentSession { this.#goalModeState = state; } + getVibeModeState(): VibeModeState | undefined { + return this.#vibeModeState; + } + + setVibeModeState(state: VibeModeState | undefined): void { + this.#vibeModeState = state; + } + get goalRuntime(): GoalRuntime { return this.#goalRuntime; } @@ -7203,6 +7214,21 @@ export class AgentSession { ); } + async sendVibeModeContext(options?: { deliverAs?: "steer" | "followUp" | "nextTurn" }): Promise { + const message = this.#buildVibeModeMessage(); + if (!message) return; + await this.sendCustomMessage( + { + customType: message.customType, + content: message.content, + display: message.display, + details: message.details, + attribution: message.attribution, + }, + options ? { deliverAs: options.deliverAs } : undefined, + ); + } + resolveRoleModel(role: string): Model | undefined { return this.#resolveRoleModelFull(role, this.#modelRegistry.getAvailable(), this.model).model; } @@ -7331,6 +7357,18 @@ export class AgentSession { }; } + #buildVibeModeMessage(): CustomMessage | null { + if (!this.#vibeModeState?.enabled) return null; + return { + role: "custom", + customType: "vibe-mode-context", + content: prompt.render(vibeModeActivePrompt), + display: false, + attribution: "agent", + timestamp: Date.now(), + }; + } + #sanitizeGoalTodoText(text: string): string { return escapeXmlText(text) .replace(/\r\n/g, "\\n") @@ -7768,6 +7806,10 @@ export class AgentSession { if (goalModeMessage) { messages.push(goalModeMessage); } + const vibeModeMessage = this.#buildVibeModeMessage(); + if (vibeModeMessage) { + messages.push(vibeModeMessage); + } if (options?.prependMessages) { messages.push(...options.prependMessages); } diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 5f9bdfc2e..2a25a4f95 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -250,6 +250,22 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ runtime.ctx.editor.setText(""); }, }, + { + name: "vibe", + description: "Toggle vibe mode (direct persistent fast/good worker sessions; read-only toolset)", + inlineHint: "[prompt]", + allowArgs: true, + getTuiAutocompleteDescription: runtime => { + if (runtime.ctx.vibeModeEnabled) return "Vibe: on"; + if (runtime.ctx.planModeEnabled) return "Vibe: blocked by plan mode"; + if (runtime.ctx.goalModeEnabled) return "Vibe: blocked by goal mode"; + return "Vibe: off"; + }, + handleTui: async (command, runtime) => { + await runtime.ctx.handleVibeModeCommand(command.args || undefined); + runtime.ctx.editor.setText(""); + }, + }, { name: "goal", description: "Toggle goal mode (persistent autonomous objective for this session)", diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 8ea30c85d..263a2d425 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -64,6 +64,7 @@ import { reportFindingTool } from "./review"; import { SearchToolBm25Tool } from "./search-tool-bm25"; import { loadSshTool } from "./ssh"; import { type TodoPhase, TodoTool } from "./todo"; +import { VibeKillTool, VibeListTool, VibeSendTool, VibeSpawnTool, VibeWaitTool } from "./vibe"; import { WriteTool } from "./write"; import { YieldTool } from "./yield"; @@ -103,6 +104,7 @@ export * from "./search-tool-bm25"; export * from "./ssh"; export * from "./todo"; export * from "./tts"; +export * from "./vibe"; export * from "./write"; export * from "./yield"; @@ -483,6 +485,11 @@ export const HIDDEN_TOOLS: Record = { report_tool_issue: s => createReportToolIssueTool(s), resolve: s => new ResolveTool(s), goal: s => new GoalTool(s), + vibe_spawn: s => new VibeSpawnTool(s), + vibe_send: s => new VibeSendTool(s), + vibe_wait: s => new VibeWaitTool(s), + vibe_kill: s => new VibeKillTool(s), + vibe_list: s => new VibeListTool(s), }; export type ToolName = BuiltinToolName; diff --git a/packages/coding-agent/src/tools/renderers.ts b/packages/coding-agent/src/tools/renderers.ts index 3e6baf824..d76bd1e9c 100644 --- a/packages/coding-agent/src/tools/renderers.ts +++ b/packages/coding-agent/src/tools/renderers.ts @@ -30,6 +30,7 @@ import { resolveToolRenderer } from "./resolve"; import { searchToolBm25Renderer } from "./search-tool-bm25"; import { sshToolRenderer } from "./ssh"; import { todoToolRenderer } from "./todo"; +import { createVibeToolRenderer } from "./vibe"; import { writeToolRenderer } from "./write"; /** @@ -111,5 +112,10 @@ export const toolRenderers: Record = { github: githubToolRenderer as ToolRenderer, goal: goalToolRenderer as ToolRenderer, web_search: webSearchToolRenderer as ToolRenderer, + vibe_spawn: createVibeToolRenderer("spawn") as ToolRenderer, + vibe_send: createVibeToolRenderer("send") as ToolRenderer, + vibe_wait: createVibeToolRenderer("wait") as ToolRenderer, + vibe_kill: createVibeToolRenderer("kill") as ToolRenderer, + vibe_list: createVibeToolRenderer("list") as ToolRenderer, write: writeToolRenderer as ToolRenderer, }; From b60cbb83b7f717deeb0c2109e8e8bad35f23451d Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 02:12:08 +0200 Subject: [PATCH 129/205] test(vibe): validated vibe runtime session lifecycle and concurrency behavior - Implemented test suite for VibeSessionRegistry covering lifecycle state management, session spawning, and turn execution. - Added verification for session steering, turn queuing, and subagent follow-up behavior within active sessions. - Validated concurrency handling in the registry including multi-session wait mechanics and result suppression. - Incorporated session cleanup logic verification for both individual kill operations and mass termination via killAll. --- .../test/vibe/vibe-runtime.test.ts | 485 ++++++++++++++++++ 1 file changed, 485 insertions(+) create mode 100644 packages/coding-agent/test/vibe/vibe-runtime.test.ts diff --git a/packages/coding-agent/test/vibe/vibe-runtime.test.ts b/packages/coding-agent/test/vibe/vibe-runtime.test.ts new file mode 100644 index 000000000..6c67ea435 --- /dev/null +++ b/packages/coding-agent/test/vibe/vibe-runtime.test.ts @@ -0,0 +1,485 @@ +/** + * Contracts: vibe worker-session registry lifecycle. + * + * 1. `spawn` returns immediately (session id + turn job id) while the turn + * runs in the background; the settled turn self-delivers a result carrying + * the activity trace AND the worker's response, and the session stays + * addressable (idle) afterwards. + * 2. `send` routes by state: steering into a streaming mid-turn worker, + * queueing when the worker is mid-turn but not steerable (drained into the + * next turn automatically), and starting a follow-up turn on the SAME + * worker id when idle. + * 3. `runSubagentFollowUpTurn` continues a live session in place: consecutive + * turns hit the same AgentSession instance (context retained) and the + * finalized result carries the yield payload + tool trace. + * 4. `wait` wakes on the FIRST settling turn among concurrent sessions and + * acknowledges its delivery so the result is not delivered twice. + * 5. `kill` cancels the in-flight turn job and releases the worker session. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async/job-manager"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { AgentLifecycleManager } from "@oh-my-pi/pi-coding-agent/registry/agent-lifecycle"; +import { AgentRegistry } from "@oh-my-pi/pi-coding-agent/registry/agent-registry"; +import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import * as executorModule from "@oh-my-pi/pi-coding-agent/task/executor"; +import type { AgentProgress, SingleResult } from "@oh-my-pi/pi-coding-agent/task/types"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { VibeSessionRegistry } from "@oh-my-pi/pi-coding-agent/vibe/runtime"; + +function createSession(options: { manager?: AsyncJobManager } = {}): ToolSession { + return { + cwd: "/tmp", + hasUI: false, + settings: Settings.isolated({}), + getSessionFile: () => null, + getSessionSpawns: () => "*", + asyncJobManager: options.manager, + } as unknown as ToolSession; +} + +function makeResult(id: string, overrides: Partial = {}): SingleResult { + return { + index: 0, + id, + agent: "task", + agentSource: "bundled", + task: "prompt", + exitCode: 0, + output: "All done.", + stderr: "", + truncated: false, + durationMs: 5, + tokens: 0, + requests: 1, + ...overrides, + }; +} + +interface Deferred { + promise: Promise; + resolve: () => void; +} + +function deferred(): Deferred { + const { promise, resolve } = Promise.withResolvers(); + return { promise, resolve }; +} + +async function pollUntil(predicate: () => boolean, timeoutMs = 2000): Promise { + const start = Date.now(); + while (!predicate()) { + if (Date.now() - start > timeoutMs) throw new Error("pollUntil timed out"); + await Bun.sleep(5); + } +} + +/** + * Minimal stand-in for a worker AgentSession: records prompts/steers, replays + * a scripted event stream through subscribed listeners on each prompt, and + * reports a final assistant message — enough surface for the executor's run + * monitor + driveSessionToYield. + */ +function createFakeWorkerSession(options: { streaming?: boolean } = {}) { + const listeners = new Set<(event: unknown) => void>(); + const prompts: string[] = []; + const steers: string[] = []; + let disposed = false; + let lastAssistant: { stopReason: string; content: Array<{ type: string; text: string }> } | undefined; + let script: { events: unknown[]; responseText: string } | undefined; + const fake = { + isStreaming: options.streaming ?? false, + model: undefined, + subscribe(listener: (event: unknown) => void): () => void { + listeners.add(listener); + return () => listeners.delete(listener); + }, + async prompt(text: string): Promise { + prompts.push(text); + const active = script; + script = undefined; + if (active) { + for (const event of active.events) { + for (const listener of [...listeners]) listener(event); + } + lastAssistant = { stopReason: "stop", content: [{ type: "text", text: active.responseText }] }; + const end = { type: "message_end", message: { role: "assistant", content: lastAssistant.content } }; + for (const listener of [...listeners]) listener(end); + } + return true; + }, + async steer(text: string): Promise { + steers.push(text); + }, + async waitForIdle(): Promise {}, + getLastAssistantMessage() { + return lastAssistant; + }, + async abort(): Promise {}, + async dispose(): Promise { + disposed = true; + }, + }; + return { + session: fake as unknown as AgentSession, + prompts, + steers, + isDisposed: () => disposed, + setStreaming(value: boolean) { + fake.isStreaming = value; + }, + setScript(next: { events: unknown[]; responseText: string }) { + script = next; + }, + }; +} + +/** Scripted turn: one `read` tool call, then a successful `yield` carrying `data`. */ +function yieldTurnEvents(data: unknown): unknown[] { + return [ + { type: "tool_execution_start", toolName: "read", args: { path: "src/foo.ts" }, intent: "Reading foo" }, + { type: "tool_execution_end", toolName: "read", result: {}, isError: false }, + { type: "tool_execution_start", toolName: "yield", args: {} }, + { + type: "tool_execution_end", + toolName: "yield", + result: { details: { status: "success", data } }, + isError: false, + }, + ]; +} + +/** Progress snapshot in the shape the executor's run monitor emits. */ +function progressSnapshot(id: string, overrides: Partial = {}): AgentProgress { + return { + index: 0, + id, + agent: "task", + agentSource: "bundled", + status: "running", + task: "prompt", + recentTools: [], + recentOutput: [], + toolCount: 0, + requests: 0, + tokens: 0, + cost: 0, + durationMs: 0, + ...overrides, + }; +} + +describe("vibe session registry", () => { + const managers: AsyncJobManager[] = []; + + function createManager(): AsyncJobManager { + const manager = new AsyncJobManager({ onJobComplete: () => {} }); + managers.push(manager); + return manager; + } + + beforeEach(() => { + AgentRegistry.resetGlobalForTests(); + AgentLifecycleManager.resetGlobalForTests(); + VibeSessionRegistry.resetGlobalForTests(); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + for (const manager of managers.splice(0)) { + await manager.dispose({ timeoutMs: 1000 }); + } + VibeSessionRegistry.resetGlobalForTests(); + AgentLifecycleManager.resetGlobalForTests(); + AgentRegistry.resetGlobalForTests(); + }); + + it("spawn returns immediately and self-delivers a turn result with activity trace + response", async () => { + const gate = deferred(); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + AgentRegistry.global().register({ + id: options.id, + displayName: options.id, + kind: "sub", + parentId: "Main", + session: createFakeWorkerSession().session, + status: "running", + }); + options.onProgress?.( + progressSnapshot(options.id, { + toolCount: 2, + recentTools: [ + { tool: "bash", args: "bun test", endMs: 2 }, + { tool: "read", args: "src/foo.ts", endMs: 1 }, + ], + lastIntent: "Running tests", + resolvedModel: "prov/fast-model", + }), + ); + await gate.promise; + AgentRegistry.global().setStatus(options.id, "idle"); + return makeResult(options.id, { output: "Implemented the widget.", requests: 3 }); + }); + + const manager = createManager(); + const session = createSession({ manager }); + const registry = VibeSessionRegistry.global(); + + const { id, jobId } = await registry.spawn(session, { cli: "fast", name: "Fast", prompt: "Build the widget." }); + expect(id).toBe("Fast"); + + // Ack is immediate: the job is still running behind the gate. + const job = manager.getJob(jobId)!; + expect(job.status).toBe("running"); + expect(registry.list("Main")[0]?.cli).toBe("fast"); + + gate.resolve(); + await job.promise; + + expect(job.status).toBe("completed"); + const text = job.resultText ?? ""; + // Envelope + summarized activity (compressed tool trace, oldest first) + response. + expect(text).toContain(' { + const gate = deferred(); + const fake = createFakeWorkerSession({ streaming: true }); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + AgentRegistry.global().register({ + id: options.id, + displayName: options.id, + kind: "sub", + parentId: "Main", + session: fake.session, + status: "running", + }); + await gate.promise; + AgentRegistry.global().setStatus(options.id, "idle"); + return makeResult(options.id); + }); + const followUps: Array<{ id: string; message: string }> = []; + vi.spyOn(executorModule, "runSubagentFollowUpTurn").mockImplementation(async options => { + followUps.push({ id: options.id, message: options.message }); + return makeResult(options.id, { output: "queued work done" }); + }); + + const manager = createManager(); + const session = createSession({ manager }); + const registry = VibeSessionRegistry.global(); + const { jobId } = await registry.spawn(session, { cli: "good", name: "Good", prompt: "Design it." }); + await pollUntil(() => AgentRegistry.global().get("Good") !== undefined); + + // Streaming worker → steering. + const steered = await registry.send(session, { session: "Good", message: "Focus on the API first." }); + expect(steered.mode).toBe("steered"); + expect(fake.steers).toEqual(["Focus on the API first."]); + + // Not streaming → queued for the next turn. + fake.setStreaming(false); + const queued = await registry.send(session, { session: "Good", message: "Then write tests." }); + expect(queued.mode).toBe("queued"); + expect(registry.list("Main")[0]?.queued).toBe(1); + + // Settling the turn drains the queue into an automatic follow-up turn. + gate.resolve(); + await manager.getJob(jobId)!.promise; + await pollUntil(() => followUps.length === 1); + expect(followUps[0]).toEqual({ id: "Good", message: "Then write tests." }); + }); + + it("send to an idle session starts a follow-up turn on the same worker", async () => { + const gate = deferred(); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + AgentRegistry.global().register({ + id: options.id, + displayName: options.id, + kind: "sub", + parentId: "Main", + session: createFakeWorkerSession().session, + status: "running", + }); + await gate.promise; + AgentRegistry.global().setStatus(options.id, "idle"); + return makeResult(options.id); + }); + const followUps: Array<{ id: string; message: string }> = []; + vi.spyOn(executorModule, "runSubagentFollowUpTurn").mockImplementation(async options => { + followUps.push({ id: options.id, message: options.message }); + options.onProgress?.( + progressSnapshot(options.id, { + toolCount: 1, + recentTools: [{ tool: "edit", args: "src/foo.ts", endMs: 1 }], + }), + ); + return makeResult(options.id, { output: "Renamed everything." }); + }); + + const manager = createManager(); + const session = createSession({ manager }); + const registry = VibeSessionRegistry.global(); + const spawn = await registry.spawn(session, { cli: "fast", name: "Fast", prompt: "First task." }); + gate.resolve(); + await manager.getJob(spawn.jobId)!.promise; + + const outcome = await registry.send(session, { session: "Fast", message: "Now rename the helpers." }); + expect(outcome.mode).toBe("turn"); + const turnJob = manager.getJob(outcome.jobId!)!; + await turnJob.promise; + + expect(followUps).toEqual([{ id: "Fast", message: "Now rename the helpers." }]); + const text = turnJob.resultText ?? ""; + expect(text).toContain('turn="2"'); + expect(text).toContain("edit(src/foo.ts)"); + expect(text).toContain("Renamed everything."); + expect(registry.list("Main")[0]?.turns).toBe(2); + }); + + it("runSubagentFollowUpTurn continues the same live session and finalizes trace + yield response", async () => { + const fake = createFakeWorkerSession(); + AgentRegistry.global().register({ + id: "Worker", + displayName: "Worker", + kind: "sub", + parentId: "Main", + session: fake.session, + status: "idle", + }); + const agent = { name: "task", description: "worker", systemPrompt: "sp", source: "bundled" as const }; + + fake.setScript({ events: yieldTurnEvents({ report: "did the first thing" }), responseText: "first summary" }); + const progressSnapshots: AgentProgress[] = []; + const first = await executorModule.runSubagentFollowUpTurn({ + id: "Worker", + agent, + message: "do the first thing", + onProgress: progress => progressSnapshots.push({ ...progress, recentTools: progress.recentTools.slice() }), + }); + expect(first.exitCode).toBe(0); + expect(first.output).toContain("did the first thing"); + expect(progressSnapshots.some(progress => progress.recentTools.some(entry => entry.tool === "read"))).toBe(true); + + // Second turn lands on the SAME session instance — prior context retained. + fake.setScript({ events: yieldTurnEvents({ report: "built on prior work" }), responseText: "second summary" }); + const second = await executorModule.runSubagentFollowUpTurn({ id: "Worker", agent, message: "now extend it" }); + expect(second.exitCode).toBe(0); + expect(second.output).toContain("built on prior work"); + expect(fake.prompts).toEqual(["do the first thing", "now extend it"]); + expect(fake.isDisposed()).toBe(false); + }); + + it("wait wakes on the first settling turn among concurrent sessions and suppresses its re-delivery", async () => { + const gates = new Map(); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + AgentRegistry.global().register({ + id: options.id, + displayName: options.id, + kind: "sub", + parentId: "Main", + session: createFakeWorkerSession().session, + status: "running", + }); + const gate = deferred(); + gates.set(options.id, gate); + await gate.promise; + AgentRegistry.global().setStatus(options.id, "idle"); + return makeResult(options.id, { output: `${options.id} finished.` }); + }); + + const manager = createManager(); + const session = createSession({ manager }); + const registry = VibeSessionRegistry.global(); + const fast = await registry.spawn(session, { cli: "fast", name: "Fast", prompt: "Task A." }); + const good = await registry.spawn(session, { cli: "good", name: "Good", prompt: "Task B." }); + await pollUntil(() => gates.size === 2); + + const waitPromise = registry.wait(session, { sessions: ["Fast", "Good"], timeoutMs: 5000 }); + gates.get("Fast")!.resolve(); + const outcome = await waitPromise; + + expect(outcome.timedOut).toBe(false); + expect(outcome.settled.map(entry => entry.id)).toEqual(["Fast"]); + expect(outcome.settled[0]!.resultText).toContain("Fast finished."); + expect(outcome.stillRunning).toEqual(["Good"]); + // The reported result must not be delivered a second time as a follow-up. + expect(manager.isDeliverySuppressed(fast.jobId)).toBe(true); + expect(manager.isDeliverySuppressed(good.jobId)).toBe(false); + + gates.get("Good")!.resolve(); + await manager.getJob(good.jobId)!.promise; + }); + + it("kill cancels the in-flight turn and releases the worker session", async () => { + const gate = deferred(); + const fake = createFakeWorkerSession(); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + AgentRegistry.global().register({ + id: options.id, + displayName: options.id, + kind: "sub", + parentId: "Main", + session: fake.session, + status: "running", + }); + await gate.promise; + return makeResult(options.id); + }); + + const manager = createManager(); + const session = createSession({ manager }); + const registry = VibeSessionRegistry.global(); + const { jobId } = await registry.spawn(session, { cli: "fast", name: "Doomed", prompt: "Never mind." }); + await pollUntil(() => AgentRegistry.global().get("Doomed") !== undefined); + + const outcome = await registry.kill(session, "Doomed"); + expect(outcome.cancelledTurn).toBe(true); + expect(manager.getJob(jobId)!.status).toBe("cancelled"); + expect(fake.isDisposed()).toBe(true); + expect(AgentRegistry.global().get("Doomed")).toBeUndefined(); + expect(registry.list("Main")[0]?.state).toBe("dead"); + await expect(registry.send(session, { session: "Doomed", message: "hello?" })).rejects.toThrow("dead"); + + gate.resolve(); + }); + + it("killAll terminates every session for the owner (mode-exit path)", async () => { + const gates = new Map(); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + AgentRegistry.global().register({ + id: options.id, + displayName: options.id, + kind: "sub", + parentId: "Main", + session: createFakeWorkerSession().session, + status: "running", + }); + const gate = deferred(); + gates.set(options.id, gate); + await gate.promise; + return makeResult(options.id); + }); + + const manager = createManager(); + const session = createSession({ manager }); + const registry = VibeSessionRegistry.global(); + await registry.spawn(session, { cli: "fast", name: "One", prompt: "A." }); + await registry.spawn(session, { cli: "good", name: "Two", prompt: "B." }); + await pollUntil(() => gates.size === 2); + + const killed = await registry.killAll("Main", manager); + expect(killed).toBe(2); + expect(registry.listIds("Main")).toEqual([]); + expect(AgentRegistry.global().get("One")).toBeUndefined(); + expect(AgentRegistry.global().get("Two")).toBeUndefined(); + + for (const gate of gates.values()) gate.resolve(); + }); +}); From 514a8ca6c8cb9995d774e1182b7adc9896c91767 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 02:12:08 +0200 Subject: [PATCH 130/205] docs(api): updated status line test helpers to include vibe mode support - Added vibeMode property to mock context in status line unit tests. - Ensured test helper functions maintain consistency with updated segment context. --- packages/coding-agent/test/status-line-model.test.ts | 1 + packages/coding-agent/test/status-line-overflow.test.ts | 1 + packages/coding-agent/test/status-line-path.test.ts | 1 + packages/coding-agent/test/status-line-time-spent.test.ts | 1 + 4 files changed, 4 insertions(+) diff --git a/packages/coding-agent/test/status-line-model.test.ts b/packages/coding-agent/test/status-line-model.test.ts index bbf8f1bcd..6a2e5ebd0 100644 --- a/packages/coding-agent/test/status-line-model.test.ts +++ b/packages/coding-agent/test/status-line-model.test.ts @@ -23,6 +23,7 @@ function createModelContext(advisorActive: boolean): SegmentContext { planMode: null, loopMode: null, goalMode: null, + vibeMode: null, collab: null, usageStats: { input: 0, diff --git a/packages/coding-agent/test/status-line-overflow.test.ts b/packages/coding-agent/test/status-line-overflow.test.ts index b73b9d230..996389989 100644 --- a/packages/coding-agent/test/status-line-overflow.test.ts +++ b/packages/coding-agent/test/status-line-overflow.test.ts @@ -46,6 +46,7 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null planMode: null, loopMode: null, goalMode: null, + vibeMode: null, collab: null, usageStats: { input: 0, diff --git a/packages/coding-agent/test/status-line-path.test.ts b/packages/coding-agent/test/status-line-path.test.ts index 587ec20bb..c32526a59 100644 --- a/packages/coding-agent/test/status-line-path.test.ts +++ b/packages/coding-agent/test/status-line-path.test.ts @@ -32,6 +32,7 @@ function createPathContext(): SegmentContext { planMode: null, loopMode: null, goalMode: null, + vibeMode: null, collab: null, usageStats: { input: 0, diff --git a/packages/coding-agent/test/status-line-time-spent.test.ts b/packages/coding-agent/test/status-line-time-spent.test.ts index 230434fcc..35620a132 100644 --- a/packages/coding-agent/test/status-line-time-spent.test.ts +++ b/packages/coding-agent/test/status-line-time-spent.test.ts @@ -42,6 +42,7 @@ function createCtx(activeMs: number): SegmentContext { planMode: null, loopMode: null, goalMode: null, + vibeMode: null, collab: null, usageStats: { input: 0, From acd893536dae8f78a36cad46fdd40c06bc954a19 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 02:27:58 +0200 Subject: [PATCH 131/205] feat(coding-agent): introduced vibe mode for persistent background agents - Implemented /vibe mode providing persistent background worker sessions and five new specialized management tools. - Developed a multi-view UI featuring a mini-composer for tool calls and a TV wall for real-time worker status, traces, and output monitoring. - Integrated shimmering effects and stable screen ordering to enhance worker state visibility and session management. - Added comprehensive unit tests for Vibe tool renderers and updated runtime registry logic to support rich screen snapshots. --- packages/coding-agent/CHANGELOG.md | 2 + .../src/tools/__tests__/vibe-render.test.ts | 173 +++++++++ packages/coding-agent/src/tools/vibe.ts | 328 ++++++++++++++---- packages/coding-agent/src/vibe/runtime.ts | 135 +++++-- .../test/vibe/vibe-runtime.test.ts | 10 +- 5 files changed, 546 insertions(+), 102 deletions(-) create mode 100644 packages/coding-agent/src/tools/__tests__/vibe-render.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a3d94abdb..f385962a5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,8 @@ ### Added +- Added live TUI "TV wall" for `/vibe` tools showing active worker activity, tool traces, and output +- Added `/vibe` mode: the model becomes a director whose toolset is stripped to `read` plus five new session tools (`vibe_spawn`, `vibe_send`, `vibe_wait`, `vibe_kill`, `vibe_list`) for driving persistent background worker sessions. Workers come in two flavors mapped to existing model tiers — `fast` (sonic/`pi/smol`) and `good` (task/`pi/task`) — retain their conversation across turns via the subagent keep-alive lifecycle, deliver each turn's result asynchronously with a compressed tool-call trace plus the worker's response, and are killed when the mode exits. The TUI renders sends as a mini CLI composer and waits as a live stacked-screen view of every worker's tool calls and streamed output. - `omp acp` now prints a short hint on stderr when launched from an interactive terminal (stdin is a TTY): the command speaks JSON-RPC over stdout and is meant to be spawned by an ACP client such as Zed, so running it by hand previously showed nothing at all ### Changed diff --git a/packages/coding-agent/src/tools/__tests__/vibe-render.test.ts b/packages/coding-agent/src/tools/__tests__/vibe-render.test.ts new file mode 100644 index 000000000..35d7f019d --- /dev/null +++ b/packages/coding-agent/src/tools/__tests__/vibe-render.test.ts @@ -0,0 +1,173 @@ +/** + * Contracts: vibe tool renderers. + * + * 1. spawn/send render a mini composer — the message typed into a tiny CLI + * frame with a prompt glyph and (while pending) a blinking cursor. + * 2. wait/list render the TV wall: one boxed screen per worker, stacked, a + * running screen showing its tool-call trace, current tool, and streamed + * text tail; an idle screen its last-activity gist; a settled screen its + * delivery footer. + * 3. Every emitted line respects the render width (sanitized, truncated). + */ +import { beforeAll, describe, expect, it } from "bun:test"; +import { Settings } from "../../config/settings"; +import { getThemeByName, setThemeInstance, type Theme } from "../../modes/theme/theme"; +import type { VibeScreenSnapshot } from "../../vibe/runtime"; +import { createVibeToolRenderer, type VibeToolDetails } from "../vibe"; + +const strip = (lines: readonly string[]): string[] => + lines.map(line => + line.replace(/\x1b\]8;[^\x1b\x07]*(?:\x07|\x1b\\)/g, "").replace(/\x1b\[[0-9;]*m/g, ""), + ); + +function makeScreen(overrides: Partial = {}): VibeScreenSnapshot { + return { + id: "Anna", + cli: "fast", + state: "running", + turns: 1, + queued: 0, + trace: [], + outputTail: [], + lastActivityAt: Date.now(), + ...overrides, + }; +} + +function renderLines(component: { render(width: number): readonly string[] }, width = 100): string[] { + return strip(component.render(width)); +} + +describe("vibe tool renderers", () => { + let uiTheme: Theme; + + beforeAll(async () => { + await Settings.init({ inMemory: true }); + const loaded = await getThemeByName("dark"); + if (!loaded) throw new Error("theme unavailable"); + uiTheme = loaded; + setThemeInstance(uiTheme); + }); + + it("send composer types the message into a mini CLI frame with a blinking cursor while pending", () => { + const renderer = createVibeToolRenderer("send"); + const component = renderer.renderCall( + { session: "Anna", message: "Focus on the API first.\nThen tests." }, + { expanded: false, isPartial: true, spinnerFrame: 0 }, + uiTheme, + ) as { render(width: number): readonly string[] }; + const text = renderLines(component).join("\n"); + + expect(text).toContain("vibe send → Anna"); + expect(text).toContain("> Focus on the API first."); + expect(text).toContain("Then tests.▌"); + expect(text).toContain("delivering…"); + // Odd frame: cursor blinks off. + const off = renderLines( + renderer.renderCall( + { session: "Anna", message: "Hi" }, + { expanded: false, isPartial: true, spinnerFrame: 1 }, + uiTheme, + ) as { render(width: number): readonly string[] }, + ).join("\n"); + expect(off).not.toContain("▌"); + }); + + it("send result frames the ack under the composer", () => { + const renderer = createVibeToolRenderer("send"); + const details: VibeToolDetails = { + op: "send", + screens: [makeScreen()], + send: { id: "Anna", mode: "steered" }, + }; + const component = renderer.renderResult( + { content: [{ type: "text", text: "ack" }], details }, + { expanded: false, isPartial: false }, + uiTheme, + { session: "Anna", message: "Focus on the API first." }, + ) as { render(width: number): readonly string[] }; + const text = renderLines(component).join("\n"); + + expect(text).toContain("vibe send → Anna"); + expect(text).toContain("> Focus on the API first."); + expect(text).toContain("steered into the running turn"); + expect(text).not.toContain("▌"); + }); + + it("wait renders stacked TV screens: live trace + streamed text, idle gist, settled footer", () => { + const renderer = createVibeToolRenderer("wait"); + const details: VibeToolDetails = { + op: "wait", + screens: [ + makeScreen({ + id: "Anna", + cli: "fast", + state: "running", + turnStartedAt: Date.now() - 5000, + turnMessage: "Build the widget", + trace: ["read(src/foo.ts)", "bash(bun test)"], + currentTool: "edit", + lastIntent: "Fixing the parser", + outputTail: ["The parser now accepts nested arrays"], + model: "prov/fast-model", + }), + makeScreen({ id: "Bob", cli: "good", state: "idle", turns: 2, lastActivity: "turn 2 completed" }), + ], + wait: { settled: [{ id: "Bob", jobId: "Bob-t2", status: "completed" }], stillRunning: ["Anna"], timedOut: false }, + }; + const component = renderer.renderResult( + { content: [{ type: "text", text: "" }], details }, + { expanded: true, isPartial: true, spinnerFrame: 2 }, + uiTheme, + { sessions: ["Anna", "Bob"] }, + ) as { render(width: number): readonly string[] }; + const lines = renderLines(component); + const text = lines.join("\n"); + + // One framed screen per worker, stacked. + expect(lines.filter(line => line.includes("╭─")).length).toBe(2); + expect(lines.filter(line => line.startsWith("╰─")).length).toBe(2); + // Live screen: header, typed turn message, trace, current tool, streamed tail. + expect(text).toContain("Anna"); + // Badge glyphs are theme-driven (⟦fast⟧ on dark); assert the flavor label itself. + expect(text).toMatch(/fast.\s*Anna/u); + expect(text).toContain("> Build the widget"); + expect(text).toContain("read(src/foo.ts)"); + expect(text).toContain("bash(bun test)"); + expect(text).toContain("edit: Fixing the parser"); + expect(text).toContain("The parser now accepts nested arrays"); + expect(text).toContain("prov/fast-model"); + // Idle screen + settled footer. + expect(text).toContain("Bob"); + expect(text).toContain("turn 2 completed"); + expect(text).toContain("turn completed — result delivered"); + // Wall header counts what is on air. + expect(text).toContain("1 on air"); + }); + + it("clamps every TV line to the render width", () => { + const renderer = createVibeToolRenderer("list"); + const details: VibeToolDetails = { + op: "list", + screens: [ + makeScreen({ + id: "VeryLongSessionNameForTruncation", + trace: [`read(${"x".repeat(200)})`], + outputTail: ["y".repeat(300)], + currentTool: "bash", + currentToolArgs: "z".repeat(200), + }), + ], + }; + const component = renderer.renderResult( + { content: [{ type: "text", text: "" }], details }, + { expanded: true, isPartial: false }, + uiTheme, + {}, + ) as { render(width: number): readonly string[] }; + const width = 48; + for (const line of renderLines(component, width)) { + expect(line.length).toBeLessThanOrEqual(width); + } + }); +}); diff --git a/packages/coding-agent/src/tools/vibe.ts b/packages/coding-agent/src/tools/vibe.ts index 8b29a8beb..3be4649df 100644 --- a/packages/coding-agent/src/tools/vibe.ts +++ b/packages/coding-agent/src/tools/vibe.ts @@ -4,13 +4,19 @@ * Five thin tools over {@link VibeSessionRegistry}: spawn/send/wait/kill/list * persistent worker sessions ("fast"/"good" CLIs). Spawns and sends return * immediately; turn results self-deliver through the async job manager. + * + * The TUI renderers lean into the "you are driving little CLIs" fiction: + * spawn/send draw a mini composer (a message typed into a tiny Claude-Code-like + * terminal), and wait/list draw the "TV wall" — one live screen per worker, + * stacked, each showing its tool calls and streamed text as it works. */ -import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import type { AgentTool, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; +import { shimmerEnabled, shimmerText } from "../modes/theme/shimmer"; import type { Theme } from "../modes/theme/theme"; import vibeKillDescription from "../prompts/tools/vibe-kill.md" with { type: "text" }; import vibeListDescription from "../prompts/tools/vibe-list.md" with { type: "text" }; @@ -23,7 +29,7 @@ import { renderStatusLine } from "../tui"; import { type VibeCli, type VibeKillOutcome, - type VibeRosterEntry, + type VibeScreenSnapshot, type VibeSendOutcome, VibeSessionRegistry, type VibeSessionState, @@ -32,6 +38,7 @@ import type { ToolSession } from "./index"; import { Ellipsis, formatBadge, + formatDuration, formatStatusIcon, replaceTabs, type ToolUIColor, @@ -70,19 +77,22 @@ type VibeOp = "spawn" | "send" | "wait" | "kill" | "list"; /** Details payload shared by every vibe tool for TUI rendering. */ export interface VibeToolDetails { op: VibeOp; - roster: VibeRosterEntry[]; + /** Live TV-wall snapshot of the owner's worker sessions at (or during) the call. */ + screens: VibeScreenSnapshot[]; spawned?: { id: string; cli: VibeCli; jobId: string }; send?: VibeSendOutcome; wait?: { settled: Array<{ id: string; jobId: string; status: "completed" | "failed" | "cancelled" }>; stillRunning: string[]; timedOut: boolean; + /** True on interim progress emissions while the wait is still blocking. */ + waiting?: boolean; }; killed?: VibeKillOutcome; } -function rosterOf(session: ToolSession): VibeRosterEntry[] { - return VibeSessionRegistry.global().list(session.getAgentId?.() ?? MAIN_AGENT_ID); +function screensOf(session: ToolSession, ids?: string[]): VibeScreenSnapshot[] { + return VibeSessionRegistry.global().screens(session.getAgentId?.() ?? MAIN_AGENT_ID, ids); } function textResult(text: string, details: VibeToolDetails): AgentToolResult { @@ -105,7 +115,7 @@ export class VibeSpawnTool implements AgentTool { readonly name = "vibe_wait"; readonly approval = "read" as const; @@ -151,15 +163,36 @@ export class VibeWaitTool implements AgentTool, ): Promise> { - const outcome = await VibeSessionRegistry.global().wait(this.session, { - sessions: params.sessions, - timeoutMs: params.timeout !== undefined ? params.timeout * 1000 : undefined, - signal, - }); + const registry = VibeSessionRegistry.global(); + // Live TV-wall frames while the wait blocks: each tick re-snapshots the + // watched workers so their tool calls and streamed text play in place. + const emitProgress = (): void => { + onUpdate?.({ + content: [{ type: "text", text: "" }], + details: { + op: "wait", + screens: screensOf(this.session, params.sessions), + wait: { settled: [], stillRunning: [], timedOut: false, waiting: true }, + }, + }); + }; + const progressTimer = onUpdate ? setInterval(emitProgress, WAIT_PROGRESS_INTERVAL_MS) : undefined; + emitProgress(); + let outcome: Awaited>; + try { + outcome = await registry.wait(this.session, { + sessions: params.sessions, + timeoutMs: params.timeout !== undefined ? params.timeout * 1000 : undefined, + signal, + }); + } finally { + clearInterval(progressTimer); + } const details: VibeToolDetails = { op: "wait", - roster: rosterOf(this.session), + screens: screensOf(this.session, params.sessions), wait: { settled: outcome.settled.map(({ id, jobId, status }) => ({ id, jobId, status })), stillRunning: outcome.stillRunning, @@ -200,10 +233,11 @@ export class VibeKillTool implements AgentTool> { const outcome = await VibeSessionRegistry.global().kill(this.session, params.session); const cancelNote = outcome.cancelledTurn ? " Its in-flight turn was cancelled." : ""; - return textResult( - `Killed session \`${outcome.id}\`.${cancelNote} Transcript remains at history://${outcome.id}.`, - { op: "kill", roster: rosterOf(this.session), killed: outcome }, - ); + return textResult(`Killed session \`${outcome.id}\`.${cancelNote} Transcript remains at history://${outcome.id}.`, { + op: "kill", + screens: screensOf(this.session), + killed: outcome, + }); } } @@ -220,19 +254,19 @@ export class VibeListTool implements AgentTool> { - const roster = rosterOf(this.session); - const details: VibeToolDetails = { op: "list", roster }; - if (roster.length === 0) { + const screens = screensOf(this.session); + const details: VibeToolDetails = { op: "list", screens }; + if (screens.length === 0) { return textResult("No vibe sessions. Spawn one with vibe_spawn.", details); } - const lines = roster.map(entry => { + const lines = screens.map(screen => { const parts = [ - `- \`${entry.id}\` [${entry.cli}] ${entry.state}`, - `${entry.turns} turn${entry.turns === 1 ? "" : "s"}`, + `- \`${screen.id}\` [${screen.cli}] ${screen.state}`, + `${screen.turns} turn${screen.turns === 1 ? "" : "s"}`, ]; - if (entry.queued > 0) parts.push(`${entry.queued} queued`); - if (entry.model) parts.push(entry.model); - if (entry.lastActivity) parts.push(`last: ${entry.lastActivity}`); + if (screen.queued > 0) parts.push(`${screen.queued} queued`); + if (screen.model) parts.push(screen.model); + if (screen.lastActivity) parts.push(`last: ${screen.lastActivity}`); return parts.join(" · "); }); return textResult(lines.join("\n"), details); @@ -240,11 +274,16 @@ export class VibeListTool implements AgentTool 0 ? `+${entry.queued}q` : ""}`); - return truncateToWidth( - `${icon} ${badge} ${uiTheme.fg("toolOutput", entry.id)} ${uiTheme.fg("dim", entry.state)} ${turns}${gist}`, - ROSTER_LINE_WIDTH, - Ellipsis.Unicode, - ); -} - interface VibeRenderArgs { cli?: VibeCli; prompt?: string; + name?: string; session?: string; message?: string; sessions?: string[]; } +/** One-line, escape-stripped fragment for embedding in a frame row. */ +function frameText(text: string, max: number): string { + return oneLineLabel(replaceTabs(text), max); +} + +/** + * Draw a left-railed mini terminal: + * ``` + * ╭─

+ * │ + * ╰─