diff --git a/.github/VOUCHED.td b/.github/VOUCHED.td deleted file mode 100644 index e8c1d6be5..000000000 --- a/.github/VOUCHED.td +++ /dev/null @@ -1,148 +0,0 @@ -# The list of vouched (or actively denounced) users for this repository. -# -# Only vouched users can open PRs here; unvouched/denounced PRs are -# auto-closed by .github/workflows/vouch-pr.yml (mitchellh/vouch check-pr). -# Issues are intentionally NOT gated here — robomp triages those. -# -# Write-access collaborators and bots are auto-allowed and need no entry. -# A denounced user ("-handle") is always blocked, even if also listed. -# -# Syntax: -# - One handle per line (without @), sorted alphabetically. -# - Optional platform prefix: `platform:username` (default platform: github). -# - Denounce by prefixing with minus: `-username` / `-platform:username`. -# - Optional free-text reason after a space following the handle. -# -# Maintainers manage this list by commenting `!vouch` / `!denounce [user]` -# on a discussion (see .github/workflows/vouch-manage.yml). -# -# Seed (2026-06-19): authors with >=2 merged PRs in the prior 6 months. -# Audit found 0 denounce-worthy actors; all reverts were maintainer -# technical rollbacks, not abuse. See the PR/commit history for provenance. - -0ttik -a-glapinski -ak4153 -any-victor -apoc -arg3t -asafmah -azais-corentin -basedcorp99 -baylee4 -belchetz -bjin -cagedbird043 -cexll -chan1103 -chuaaron -ckumar1 -cyjaysong -daandden -daaximus -danzaio -darkphilosophy -deprecatedluke -djdembeck -dmarsh-gusto -dylanbohlender -elikoga -enieuwy -fettpl -flare576 -foreveryoungpp -fryuni -gratefuldave -h4vc -habibpro1999 -handlecusion -haosenwang1018 -heyitsgilbert -hezhiyang2000 -hpost -iacore -igasmi -infernix -inprealpha -insodimension -itertea -itzrnvr -jaaneek -jagravnaik -jasonw22 -jchristman -jdavv -jeffscottward -jiwangyihao -joswha -kamafozilov -kamijotoma -kenmege -korri123 -kukkerem -lance0 -larkinwc -ldx -lederniermagicien -loftiskg -lyc-aon -m3ridian-zero -makomakogo -masonc15 -mathews-tom -mattwilkinsonn -maximhar -maxvisionai -metaphorics -mikeei -mokto -moutazhaq -mouyase -mq1n -muness -nnk97 -nszceta -ogrodev -oldschoola -paralin -parsifa1 -pgupta-git -phanthh -pidevxplay -pppobear -qfrtt -ravshansbox -rburketaylor -renstillmann -riverpilot -romanalexander -rznmkx -scarthread -segmentationf4u1t -serverinspector -shoucandanghehe -shyndman -sit -slact -smileynet -sundbp -superhedge22 -tbui17 -tc97222 -tdiant -tjboudreaux -tsagi2045 -turbomolli -unravl -usr-bin-roygbiv -vmcall -voidchecksum -voiys -watzon -wodenjay -wolfiesch -xaviergmail -zakhar-kogan -zamorakpds -zekdevs -zommiommy diff --git a/.github/actions/build-native/action.yml b/.github/actions/build-native/action.yml index 3d778a106..7a0da342a 100644 --- a/.github/actions/build-native/action.yml +++ b/.github/actions/build-native/action.yml @@ -34,6 +34,10 @@ inputs: cross-arch linux build, or set alone for a host-arch (x64) linux build. required: false default: "" + libc: + description: Optional Linux libc artifact qualifier (for example, musl) + required: false + default: "" rust_checks: description: Run clippy/rustfmt checks (only one matrix entry should set this) required: false @@ -135,7 +139,7 @@ runs: # --- Rust flags (shared) ------------------------------------------------ - name: Configure native Rust flags - if: inputs.target == '' + if: inputs.target == '' || inputs.arch == 'x64' shell: bash env: TARGET_ARCH: ${{ inputs.arch }} @@ -178,7 +182,7 @@ runs: if: steps.detect.outputs.on_infra == 'false' uses: Swatinem/rust-cache@v2 with: - shared-key: native-${{ inputs.platform }}-${{ inputs.arch }}-${{ inputs.variant || 'default' }}-h${{ inputs.hash }} + shared-key: native-${{ inputs.platform }}-${{ inputs.libc || 'default' }}-${{ inputs.arch }}-${{ inputs.variant || 'default' }}-h${{ inputs.hash }} cache-on-failure: true save-if: ${{ inputs.save_cache == 'true' }} cache-workspace-crates: true @@ -305,7 +309,7 @@ runs: - name: Upload native addon(s) uses: actions/upload-artifact@v4 with: - name: pi-natives-${{ inputs.platform }}-${{ inputs.arch }}${{ inputs.variant && format('-{0}', inputs.variant) || '' }}-h${{ inputs.hash }} + name: pi-natives-${{ inputs.platform }}-${{ inputs.libc && format('{0}-', inputs.libc) || '' }}${{ inputs.arch }}${{ inputs.variant && format('-{0}', inputs.variant) || '' }}-h${{ inputs.hash }} path: packages/natives/native/pi_natives.${{ inputs.platform }}-${{ inputs.arch }}*.node if-no-files-found: error # Explicit so the native_artifact_lookup canary keeps working even if diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 00df0a797..c8aa4aa15 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -3,14 +3,12 @@ name: CI on: push: branches: [main] - # Vouch bookkeeping commits (mitchellh/vouch writes VOUCHED.td back to - # main on !vouch/!denounce/!unvouch) only edit the vouch list and need no - # build. Skip CI when a main push changes nothing but the vouch file; a - # push that also touches anything else still runs the full matrix. - paths-ignore: - - .github/VOUCHED.td + paths: + - "packages/**" pull_request: branches: [main] + paths: + - "packages/**" workflow_dispatch: inputs: skip_npm: @@ -144,6 +142,8 @@ jobs: # `actions/upload-artifact` `name:` template in build-native action. cross_platform_required=( "pi-natives-linux-arm64-h${hash}" + "pi-natives-linux-musl-x64-baseline-h${hash}" + "pi-natives-linux-musl-arm64-h${hash}" "pi-natives-darwin-x64-baseline-h${hash}" "pi-natives-darwin-arm64-h${hash}" "pi-natives-win32-x64-baseline-h${hash}" @@ -238,7 +238,7 @@ jobs: # building the artifacts that ship in releases. Skipped on main when # native_artifact_lookup already found a recent run with all artifacts intact. native_cross_platform_kata: - name: "Native: ${{ matrix.platform }} ${{ matrix.arch }}" + name: "Native: ${{ matrix.platform }} ${{ matrix.libc || '' }} ${{ matrix.arch }}" needs: [release_metadata, native_artifact_lookup] if: ${{ needs.release_metadata.outputs.is-release == 'true' || (github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.native_artifact_lookup.outputs.cross-platform-run-id == '') }} strategy: @@ -246,6 +246,8 @@ jobs: matrix: include: - { os: omp-kata, platform: linux, arch: arm64, target: aarch64-unknown-linux-gnu } + - { os: omp-kata, platform: linux, libc: musl, arch: x64, target: x86_64-unknown-linux-musl, variant: baseline } + - { os: omp-kata, platform: linux, libc: musl, arch: arm64, target: aarch64-unknown-linux-musl } - { os: omp-kata, platform: win32, arch: x64, target: x86_64-pc-windows-msvc, variant: baseline } runs-on: ${{ matrix.os }} steps: @@ -255,9 +257,10 @@ jobs: hash: ${{ needs.native_artifact_lookup.outputs.source-hash }} platform: ${{ matrix.platform }} arch: ${{ matrix.arch }} + libc: ${{ matrix.libc }} variant: ${{ matrix.variant }} target: ${{ matrix.target }} - glibc: ${{ matrix.platform == 'linux' && env.GLIBC_FLOOR || '' }} + glibc: ${{ matrix.platform == 'linux' && matrix.libc != 'musl' && env.GLIBC_FLOOR || '' }} save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }} native_cross_platform_macos: @@ -575,6 +578,16 @@ jobs: arch: x64, target_id: linux-x64, binary_path: packages/coding-agent/binaries/omp-linux-x64, + native_artifact_pattern: pi-natives-linux-x64-*, + } + - { + os: ubuntu-22.04, + platform: linux, + libc: musl, + arch: x64, + target_id: linux-musl-x64, + binary_path: packages/coding-agent/binaries/omp-linux-musl-x64, + native_artifact_pattern: pi-natives-linux-musl-x64-*, } - { os: ubuntu-24.04-arm, @@ -582,6 +595,16 @@ jobs: arch: arm64, target_id: linux-arm64, binary_path: packages/coding-agent/binaries/omp-linux-arm64, + native_artifact_pattern: pi-natives-linux-arm64*, + } + - { + os: ubuntu-24.04-arm, + platform: linux, + libc: musl, + arch: arm64, + target_id: linux-musl-arm64, + binary_path: packages/coding-agent/binaries/omp-linux-musl-arm64, + native_artifact_pattern: pi-natives-linux-musl-arm64*, } - { os: macos-15-intel, @@ -589,6 +612,7 @@ jobs: arch: x64, target_id: darwin-x64, binary_path: packages/coding-agent/binaries/omp-darwin-x64, + native_artifact_pattern: pi-natives-darwin-x64*, } - { os: macos-14, @@ -596,6 +620,7 @@ jobs: arch: arm64, target_id: darwin-arm64, binary_path: packages/coding-agent/binaries/omp-darwin-arm64, + native_artifact_pattern: pi-natives-darwin-arm64*, } - { os: ubuntu-22.04, @@ -603,6 +628,7 @@ jobs: arch: x64, target_id: win32-x64, binary_path: packages/coding-agent/binaries/omp-windows-x64.exe, + native_artifact_pattern: pi-natives-win32-x64*, } runs-on: ${{ matrix.os }} permissions: @@ -636,7 +662,7 @@ jobs: - name: Download native addon(s) uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: - pattern: pi-natives-${{ matrix.platform }}-${{ matrix.arch }}*-h${{ needs.native_artifact_lookup.outputs.source-hash }} + pattern: ${{ matrix.native_artifact_pattern }}-h${{ needs.native_artifact_lookup.outputs.source-hash }} path: packages/natives/native merge-multiple: true - name: Build release binary @@ -659,15 +685,29 @@ jobs: # Windows binary is cross-built on Linux, so we have no Windows runner # to smoke it on. Cross-build correctness is verified via the napi # entry-point exports (see build-native action) and the bun - # `--compile --target=bun-windows-x64-*` cross-compile. + # `--compile --target=bun-windows-x64-*` cross-compile. Musl binaries + # need the musl loader, which glibc runners lack — they are smoked in + # the Alpine container step below instead. - name: Smoke release binary - if: matrix.platform != 'win32' + if: matrix.platform != 'win32' && matrix.libc != 'musl' run: | runtime_dir="$(mktemp -d)" HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" "${{ matrix.binary_path }}" --version HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" "${{ matrix.binary_path }}" --smoke-test + - name: Smoke musl release binary on Alpine + if: matrix.libc == 'musl' + run: | + binary="$(realpath "${{ matrix.binary_path }}")" + # Bun's musl-target binaries link libstdc++/libgcc dynamically; + # Alpine users install them alongside the binary (same as bun itself). + docker run --rm -v "$binary:/usr/local/bin/omp:ro" alpine:3.22 sh -ec ' + apk add --no-cache libstdc++ libgcc >/dev/null + runtime_dir="$(mktemp -d)" + HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" omp --version + HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" omp --smoke-test + ' - name: Publish native addon package - if: ${{ !inputs.skip_npm }} + if: ${{ !inputs.skip_npm && matrix.libc != 'musl' }} env: # Fallback auth: setup-node wrote an .npmrc referencing # NODE_AUTH_TOKEN; npm uses it only when OIDC has no trusted diff --git a/.github/workflows/vouch-manage.yml b/.github/workflows/vouch-manage.yml deleted file mode 100644 index e7ce41b11..000000000 --- a/.github/workflows/vouch-manage.yml +++ /dev/null @@ -1,44 +0,0 @@ -name: Vouch (manage) - -# Let maintainers vouch/denounce/unvouch by commenting on a Discussion: -# !vouch vouch the discussion author -# !vouch @user [reason] vouch a specific user -# !denounce [@user] [reason] -# !unvouch [@user] -# Only collaborators with admin/maintain/write are honored (triage EXCLUDED; -# upstream's default `roles` includes triage, which we override below). -# -# Commits the VOUCHED.td change back to the default branch using the stock -# GITHUB_TOKEN (no GitHub App needed). NOTE: this works only while the default -# branch is UNPROTECTED — GITHUB_TOKEN cannot bypass branch protection. If you -# protect the branch later, switch back to a GitHub App token on a bypass list. - -on: - discussion_comment: - types: [created] - -# Serialize writes to VOUCHED.td so concurrent vouches don't clobber. -concurrency: - group: vouch-manage - cancel-in-progress: false - -permissions: - contents: write # commit VOUCHED.td - discussions: write # read the comment / acknowledge - -jobs: - manage: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - - uses: mitchellh/vouch/action/manage-by-discussion@v1 - with: - discussion-number: ${{ github.event.discussion.number }} - comment-node-id: ${{ github.event.comment.node_id }} - vouch-keyword: "!vouch" - denounce-keyword: "!denounce" - unvouch-keyword: "!unvouch" - roles: admin,maintain,write - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/vouch-pr.yml b/.github/workflows/vouch-pr.yml deleted file mode 100644 index 6afa86cdc..000000000 --- a/.github/workflows/vouch-pr.yml +++ /dev/null @@ -1,51 +0,0 @@ -name: Vouch (PR gate) - -# Auto-close PRs from unvouched or denounced users. Issues are left alone -# (robomp triages those). Runs under `pull_request_target` so the token can -# act on fork PRs; this job does NO checkout and runs NO PR code — it only -# reads .github/VOUCHED.td from the base repo and calls the GitHub API. - -on: - pull_request_target: - types: [opened, reopened, ready_for_review] - -permissions: - contents: read # read VOUCHED.td from the base branch - pull-requests: write # close + comment - issues: write # add the `vouched` label (labels use the Issues API) - -concurrency: - group: vouch-pr-${{ github.event.pull_request.number }} - cancel-in-progress: true - -jobs: - check: - runs-on: ubuntu-latest - steps: - - id: vouch - uses: mitchellh/vouch/action/check-pr@v1 - with: - pr-number: ${{ github.event.pull_request.number }} - auto-close: true - require-vouch: true # block unvouched, not only denounced - # vouched-file: .github/VOUCHED.td (default) - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - - # Survivors of the gate (vouched, or auto-allowed collaborators/bots) get - # a FRESH `vouched` label on every (re)open / ready-for-review. robomp - # reviews ONLY on that label event (ROBOMP_PR_REVIEW_TRIGGER=vouched_label), - # so review is always triggered by a just-validated PR, never a stale label. - - name: Label vouched PRs for robomp review - if: ${{ steps.vouch.outputs.status == 'vouched' || steps.vouch.outputs.status == 'allowed' }} - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - REPO: ${{ github.repository }} - PR: ${{ github.event.pull_request.number }} - run: | - gh label create vouched --repo "$REPO" --color 2da44e --description "Passed the vouch gate" --force - # remove+add so a fresh `labeled` event fires even when the label - # persisted across close/reopen (re-adding an existing label emits no - # event). The check above just re-validated, so trust is never stale. - gh pr edit "$PR" --repo "$REPO" --remove-label vouched || true - gh pr edit "$PR" --repo "$REPO" --add-label vouched diff --git a/.gitignore b/.gitignore index 552e6a3b6..db243d087 100644 --- a/.gitignore +++ b/.gitignore @@ -1,5 +1,5 @@ # Dependencies -node_modules/ +node_modules .npm/ # Build output diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 66f25b399..b6f303384 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,54 +1,90 @@ # Contributing to oh-my-pi -Thanks for your interest in contributing. This project uses a lightweight -**vouch** system to decide who can open pull requests. Please read this before -opening a PR. +Pull requests are welcome. Keep them focused, understand the work you submit, +and be prepared to explain and maintain it. -## TL;DR +> [!NOTE] +> Pull requests are **temporarily open to everyone** as a trial. We previously +> required a vouch before accepting PRs; that requirement is lifted for now +> while we evaluate how open contributions go. Depending on the results, the +> vouch system may return. -- **Issues are open to everyone.** File bugs, feature requests, and questions - freely — they are triaged automatically. -- **Pull requests require a vouch.** A PR whose author is not vouched (or is - denounced) is **closed automatically**. If you are not yet vouched, do **not** - open a PR to get noticed — it will be closed on sight. Start a Discussion and - ask to be vouched first (see below). +## Before you start -## Who can open PRs +### Small changes -A pull request is accepted when its author is any of: +Bug fixes, documentation updates, and narrowly scoped improvements can go +straight to a pull request. -- a repository collaborator (write access or above), or a bot; or -- listed — without a leading `-` — in [`.github/VOUCHED.td`](.github/VOUCHED.td). +### Major changes -Anyone **denounced** (prefixed with `-` in that file) is always blocked. +Discuss major features and broad architectural or behavioral changes in +[Discord](https://discord.gg/4NMW9cdXZa) **before writing the implementation**. +This includes new subsystems, large UI changes, new dependencies, and changes +that span several packages. A GitHub issue is not a substitute for this +discussion, and prior discussion does not guarantee that a pull request will be +merged. -## Getting vouched +### Do not open an issue for work you are about to submit -1. Open a [Discussion](../../discussions) (or comment on an existing one) - describing what you'd like to contribute. -2. A maintainer vouches you by commenting **`!vouch`** (vouches the discussion - author) or **`!vouch @your-handle`** on that discussion. -3. Once you appear in `.github/VOUCHED.td`, open your PR — it stays open and is - reviewed. +If you intend to implement a change yourself, **do not create an issue for it +first**. robomp treats actionable issues as work to pick up and may start the +same fix in parallel, wasting compute and maintainer time. -Maintainers may also `!denounce [@user]` and `!unvouch [@user]`. Only -collaborators with admin/maintain/write can run these commands. +Open an issue when you are reporting a problem or proposing work that you are +not already turning into a pull request. If a relevant issue already exists, +link it from your pull request instead of creating another one. -## What happens to your PR +## AI-assisted contributions -| You are… | Result | -| --- | --- | -| Vouched (or a collaborator) | PR stays open → automated review → human review | -| Not vouched | PR closed with a comment — get vouched, then reopen or open a new PR | -| Denounced | PR closed | +AI agents are welcome as tools, not as unattended contributors. Do not give an +agent a vague goal and submit whatever it produces. -Pushing more commits to an open, vouched PR is fine — it remains vouched. +Before opening a pull request, you must: -## The VOUCHED.td file +- constrain the agent to the agreed scope and reject unrelated changes; +- review every changed file and understand the resulting behavior; +- run the relevant checks and exercise the changed behavior yourself; and +- submit the pull request only after that review, rather than letting an agent + publish it autonomously. -[`.github/VOUCHED.td`](.github/VOUCHED.td) is the source of truth: one handle per -line, sorted alphabetically, optionally `platform:handle`, with `-` marking a -denouncement and an optional reason after the handle. The format follows -[mitchellh/vouch](https://github.com/mitchellh/vouch); the denouncement list is -intentionally public so other projects can reuse our prior knowledge of bad -actors. +You are responsible for the code, regardless of who or what generated it. + +## Pull request requirements + +Every pull request body **MUST include at least one sentence written by you, in +your own words**, explaining what changed and why. A generated summary, pasted +agent transcript, or checklist alone does not satisfy this requirement. + +One honest line is enough: + +> I reviewed the full diff; this change fixes duplicate PR reviews by reusing +> the existing delivery guard. + +You **MUST verify that the change works as intended**. `bun check` and automated +tests are expected where relevant, but they are not proof that the behavior +works. Exercise the changed path yourself and report the exact scenario and +result in the pull request: + +- for a bug fix, reproduce the bug and confirm the same reproduction no longer + fails; +- for a feature, launch the product and use the feature end to end; and +- for a UI change, interact with it and inspect the rendered result. + +“`bun check` passes” by itself is not sufficient verification. For coding-agent +development commands and repository structure, see +[`packages/coding-agent/DEVELOPMENT.md`](packages/coding-agent/DEVELOPMENT.md). + +Keep each pull request to one logical change. Avoid unrelated cleanup, +drive-by refactors, generated noise, or features that were not part of the +agreed scope. + +## Review + +Maintainers review the submitted behavior and the contributor's understanding +of it—not the volume of generated code. Respond to review feedback yourself, +and only apply suggestions you have checked. + +Pull requests may be closed when they skip required prior discussion, lack the +human-written explanation, contain unreviewed agent output, or mix unrelated +changes. diff --git a/Cargo.lock b/Cargo.lock index 14652b375..2a7fe38a3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -107,9 +107,9 @@ dependencies = [ [[package]] name = "anyhow" -version = "1.0.103" +version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2a4385e2e34eb35d6b3efe798b9eb88096925d87726c0798709bf56d9ed84af3" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" [[package]] name = "arboard" @@ -172,7 +172,7 @@ checksum = "057ae90e7256ebf85f840b1638268df0142c9d19467d500b790631fd301acc27" dependencies = [ "bit-set", "regex", - "thiserror 2.0.18", + "thiserror 2.0.19", "tree-sitter", ] @@ -193,18 +193,18 @@ checksum = "3b43422f69d8ff38f95f1b2bb76517c91589a924d1559a0e935d7c8ce0274c11" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.91" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "ae36dc4177970ef04fde5178d3e2429882def40e57a451f919c098f72baa6cec" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.3", ] [[package]] @@ -283,9 +283,9 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" [[package]] name = "bitflags" -version = "2.13.0" +version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b4388bee8683e3d04af747c73422af53102d2bd24d9eadb6cbc100baef4b43f8" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" [[package]] name = "bitvec" @@ -364,7 +364,7 @@ dependencies = [ "proc-macro2", "quote", "rustversion", - "syn", + "syn 2.0.119", ] [[package]] @@ -384,7 +384,7 @@ dependencies = [ "rlimit", "strum", "strum_macros", - "thiserror 2.0.18", + "thiserror 2.0.19", "tokio", "tracing", "uucore 0.8.0", @@ -419,7 +419,7 @@ dependencies = [ "strum", "strum_macros", "terminfo", - "thiserror 2.0.18", + "thiserror 2.0.19", "tokio", "tokio-util", "tracing", @@ -439,7 +439,7 @@ dependencies = [ "cached 0.56.0", "indenter", "peg", - "thiserror 2.0.18", + "thiserror 2.0.19", "tracing", "utf8-chars", ] @@ -456,7 +456,7 @@ dependencies = [ "indenter", "insta", "peg", - "thiserror 2.0.18", + "thiserror 2.0.19", "tracing", "utf8-chars", "uuid", @@ -493,9 +493,9 @@ checksum = "175812e0be2bccb6abe50bb8d566126198344f707e304f45c648fd8f2cc0365e" [[package]] name = "bytemuck" -version = "1.25.1" +version = "1.25.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d6aedf8ae72766347502cf3cb4f41cf5e9cc37d28bee90f1fdaaae15f9cf9424" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" dependencies = [ "bytemuck_derive", ] @@ -508,7 +508,7 @@ checksum = "f65693059b6b9c588b9f62fed1cedbf0a8b805631457ea162d68f0de186f3de5" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -534,7 +534,7 @@ dependencies = [ "cached_proc_macro_types", "hashbrown 0.15.5", "once_cell", - "thiserror 2.0.18", + "thiserror 2.0.19", "web-time", ] @@ -550,7 +550,7 @@ dependencies = [ "hashbrown 0.16.1", "once_cell", "parking_lot", - "thiserror 2.0.18", + "thiserror 2.0.19", "web-time", ] @@ -563,7 +563,7 @@ dependencies = [ "darling 0.20.11", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -575,7 +575,7 @@ dependencies = [ "darling 0.20.11", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -620,9 +620,9 @@ checksum = "fd16c4719339c4530435d38e511904438d07cce7950afa3718a84ac36c10e89e" [[package]] name = "cfg_aliases" -version = "0.2.1" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" [[package]] name = "chacha20" @@ -669,9 +669,9 @@ dependencies = [ [[package]] name = "clap" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" +checksum = "d91e0c145792ef73a6ad36d27c75ac09f1832222a3c209689d90f534685ee5b7" dependencies = [ "clap_builder", "clap_derive", @@ -679,9 +679,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.0" +version = "4.6.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" +checksum = "f09628afdcc538b57f3c6341e9c8e9970f18e4a481690a64974d7023bd33548b" dependencies = [ "anstream", "anstyle", @@ -692,14 +692,14 @@ dependencies = [ [[package]] name = "clap_derive" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ "heck", "proc-macro2", "quote", - "syn", + "syn 3.0.3", ] [[package]] @@ -741,7 +741,7 @@ dependencies = [ "nom 7.1.3", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -763,7 +763,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1b60b5124979fccd9addd89d8b97a1d6eebb4950694520c75ddd722535ea443f" dependencies = [ "nix 0.31.3", - "thiserror 2.0.18", + "thiserror 2.0.19", ] [[package]] @@ -929,9 +929,9 @@ dependencies = [ [[package]] name = "ctor" -version = "1.0.8" +version = "1.0.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fb22e947478ccf9dc44d8922042c677a63fbb88f2cb468521d1145816e5087cb" +checksum = "e2e30e509674ef0ec91e21a7735766db37d163d46151b6a361d8b83dd79116bd" [[package]] name = "darling" @@ -964,7 +964,7 @@ dependencies = [ "proc-macro2", "quote", "strsim", - "syn", + "syn 2.0.119", ] [[package]] @@ -977,7 +977,7 @@ dependencies = [ "proc-macro2", "quote", "strsim", - "syn", + "syn 2.0.119", ] [[package]] @@ -988,7 +988,7 @@ checksum = "fc34b93ccb385b40dc71c6fceac4b2ad23662c7eeb248cf10d529b7e055b6ead" dependencies = [ "darling_core 0.20.11", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -999,7 +999,7 @@ checksum = "ac3984ec7bd6cfa798e62b4a642426a5be0e68f9401cfc2a01e3fa9ea2fcdb8d" dependencies = [ "darling_core 0.23.0", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -1039,7 +1039,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ccc2776f0c61eca1ca32528f85548abd1a4be8fb53d1b21c013e4f18da1e7090" dependencies = [ "data-encoding", - "syn", + "syn 2.0.119", ] [[package]] @@ -1061,7 +1061,7 @@ dependencies = [ "defmt-parser", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -1070,7 +1070,7 @@ version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" dependencies = [ - "thiserror 2.0.18", + "thiserror 2.0.19", ] [[package]] @@ -1100,7 +1100,7 @@ version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1e0e367e4e7da84520dedcac1901e4da967309406d1e51017ae1abfb97adbd38" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "objc2", ] @@ -1112,7 +1112,7 @@ checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -1249,9 +1249,9 @@ checksum = "dd2e7510819d6fbf51a5545c8f922716ecfb14df168a3242f7d33e0239efe6a1" [[package]] name = "fastrand" -version = "2.4.1" +version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "fax" @@ -1364,7 +1364,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "54f0d287c53ffd184d04d8677f590f4ac5379785529e5e08b1c8083acdd5c198" dependencies = [ "memchr", - "thiserror 2.0.18", + "thiserror 2.0.19", ] [[package]] @@ -1429,9 +1429,9 @@ checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" [[package]] name = "futures" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "a88cf1f829d945f548cf8fec32c61b1f202b6d93b45848602fc02af4b12ad218" dependencies = [ "futures-channel", "futures-core", @@ -1444,9 +1444,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" dependencies = [ "futures-core", "futures-sink", @@ -1454,15 +1454,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "6754879cc9f2c66f88c6e5c35344bb0bdb0708b0352b1201815667c7eabc7458" dependencies = [ "futures-core", "futures-task", @@ -1471,32 +1471,32 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "4577ecaa3c4f96589d473f679a71b596316f6641bc350038b962a5daf0085d7a" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "2d6d3cde68c518367be28956066ddfef33813991b77a55005a69dae04bf3b10b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "e34418ac499d6305c2fb5ad0ed2f6ac998c5f8ca209b4510f7f94242c647e307" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" [[package]] name = "futures-timer" @@ -1506,9 +1506,9 @@ checksum = "af43fadb8a98512d547e37b4e92e0ced13e205c061b87b4623eff01d918d6968" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" dependencies = [ "futures-channel", "futures-core", @@ -1592,9 +1592,9 @@ dependencies = [ [[package]] name = "glob" -version = "0.3.3" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280" +checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" [[package]] name = "globset" @@ -1780,7 +1780,7 @@ dependencies = [ "lru", "once_cell", "regex", - "thiserror 2.0.18", + "thiserror 2.0.19", ] [[package]] @@ -2097,7 +2097,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "85518b9086bf01117761b90e7691c0ef3236fa8adfb1fb44dd248fe5f87215d5" dependencies = [ "quantette", - "thiserror 2.0.18", + "thiserror 2.0.19", ] [[package]] @@ -2108,9 +2108,9 @@ checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" [[package]] name = "ignore" -version = "0.4.29" +version = "0.4.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4ffa3a0547a138e59ddd6fa3b7c672ed47e6ad6a3cd177984ff1116aa5ba742" +checksum = "7f8a7b8211e695a1d0cd91cace480d4d0bd57667ab10277cc412c5f7f4884f83" dependencies = [ "crossbeam-deque", "globset", @@ -2182,29 +2182,29 @@ dependencies = [ [[package]] name = "inferno" -version = "0.12.7" +version = "0.12.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2c05b9ae050366b3363927f59d2c07f922a746e97e2e96f70d479a3c7dbbbe5" +checksum = "0c460d4fa06223667240720ab69a8045133755ae6dfbe100cf481b95e3a014f1" dependencies = [ "ahash", "itoa", "log", "num-format", "once_cell", - "quick-xml 0.41.0", + "quick-xml", "rgb", "str_stack", ] [[package]] name = "inherent" -version = "1.0.13" +version = "1.0.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c727f80bfa4a6c6e2508d2f05b6f4bfce242030bd88ed15ae5331c5b5d30fba7" +checksum = "bee2c455ca60511a054699102d40ce7153621cb2429558f9eaa600e4499b5984" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.3", ] [[package]] @@ -2213,7 +2213,7 @@ version = "0.11.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "inotify-sys", "libc", ] @@ -2341,11 +2341,12 @@ dependencies = [ [[package]] name = "jiff" -version = "0.2.32" +version = "0.2.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "961d16382652bfdd8c6f68b223b26a8c93e0d475c672f414411db31c6c5c900e" +checksum = "e184d09547b80eb7e20d141ba2fb1fbac843ca53f4cf1b31210adc4c1adc6e16" dependencies = [ "defmt", + "jiff-core", "jiff-static", "jiff-tzdb-platform", "log", @@ -2355,6 +2356,15 @@ dependencies = [ "windows-link", ] +[[package]] +name = "jiff-core" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" +dependencies = [ + "defmt", +] + [[package]] name = "jiff-icu" version = "0.2.2" @@ -2368,13 +2378,14 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.32" +version = "0.2.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0879bd39df99c4c5e2c6615ccc026391a423dde10532c573e6086eb94a802cc" +checksum = "323da076b7a6faf914dc677cb05a4b907742ff7375c8322c9e7f5061e5e0e9de" dependencies = [ + "jiff-core", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2438,7 +2449,7 @@ version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "07293a4e297ac234359b510362495713f75ea345d5307140414f20c69ffeb087" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "libc", ] @@ -2450,9 +2461,9 @@ checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" [[package]] name = "libc" -version = "0.2.186" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "libloading" @@ -2616,11 +2627,11 @@ dependencies = [ [[package]] name = "napi" -version = "3.10.5" +version = "3.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6826e5ddc15589b2d68c8ad5321c18e85d40488e93e32962f362e572669bccf6" +checksum = "de33522036981030a75c231829566bc63414e08101a6f5ff4ac6cef19c8e0941" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "ctor", "futures", "napi-build", @@ -2638,36 +2649,36 @@ checksum = "c9c366d2c8c60b86fa632df75f745509b52f9128f91a6bad4c796e44abb505e1" [[package]] name = "napi-derive" -version = "3.5.10" +version = "3.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0fe526e81c105d3640516fcde83909dd1afe757c0d7a15af58830b5bc0fb9a1" +checksum = "a49c513341a61a16a10af6efcce46b30d0822ba2d4fb197d24d33dfc199c78d5" dependencies = [ "convert_case", "ctor", "napi-derive-backend", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "napi-derive-backend" -version = "5.1.2" +version = "6.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "514281397bcddd9ea9a876c7a21a57bff2374237a000ca9a64ea0211ec1993e2" +checksum = "4747005fa3e2c9989ac45a723a514c5db2411238b72981a3cda4c701a9dfea17" dependencies = [ "convert_case", "proc-macro2", "quote", "semver", - "syn", + "syn 2.0.119", ] [[package]] name = "napi-sys" -version = "3.2.3" +version = "3.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "73e43cf2eb0bd1bf95a43c07c076ebd2da5d1e015a71c3d201faeffffcc0ecac" +checksum = "85fbf1fa9f1babfe396d74bbbf52b3643770243e8f5b0b46715d4caf7f0dfc9a" dependencies = [ "libloading", ] @@ -2695,7 +2706,7 @@ version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ab2156c4fce2f8df6c499cc1c763e4394b7482525bf2a9701c9d79d215f519e4" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "cfg-if", "cfg_aliases 0.1.1", "libc", @@ -2707,9 +2718,9 @@ version = "0.29.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "71e2746dc3a24dd78b3cfcb7be93368c6de9963d30f43a6a73998a9cf4b17b46" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "cfg-if", - "cfg_aliases 0.2.1", + "cfg_aliases 0.2.2", "libc", ] @@ -2719,9 +2730,9 @@ version = "0.31.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "cfg-if", - "cfg_aliases 0.2.1", + "cfg_aliases 0.2.2", "libc", ] @@ -2762,7 +2773,7 @@ version = "8.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4d3d07927151ff8575b7087f245456e549fea62edf0ec4e565a5ee50c8402bc3" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "fsevent-sys", "inotify", "kqueue", @@ -2780,7 +2791,7 @@ version = "2.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "42b8cfee0e339a0337359f3c88165702ac6e600dc01c0cc9579a92d62b08477a" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", ] [[package]] @@ -2852,7 +2863,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d49e936b501e5c5bf01fda3a9452ff86dc3ea98ad5f283e1455153142d97518c" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "objc2", "objc2-core-graphics", "objc2-foundation", @@ -2864,7 +2875,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "dispatch2", "objc2", ] @@ -2875,7 +2886,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e022c9d066895efa1345f8e33e584b9f958da2fd4cd116792e15e07e4720a807" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "dispatch2", "objc2", "objc2-core-foundation", @@ -2894,7 +2905,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e3e0adef53c21f888deb4fa59fc59f7eb17404926ee8a6f59f5df0fd7f9f3272" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "objc2", "objc2-core-foundation", ] @@ -2905,7 +2916,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "180788110936d59bab6bd83b6060ffdfffb3b922ba1396b312ae795e1de9d81d" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "objc2", "objc2-core-foundation", ] @@ -2937,7 +2948,7 @@ version = "6.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0cc3cbf698f9438986c11a880c90a6d04b9de27575afd28bbf45b154b6c709e2" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "libc", "once_cell", "onig_sys", @@ -3008,7 +3019,7 @@ dependencies = [ "by_address", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3102,9 +3113,9 @@ checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" [[package]] name = "pest" -version = "2.8.7" +version = "2.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47627dd7305c6a2d6c8c6bcd24c5a4c17dbbf425f4f9c5313e724b38fc9782e9" +checksum = "7df728be843c7070fab6ab7c328c4e9e9d78e23bf749c0669c86ee7ebfa050a2" dependencies = [ "memchr", "ucd-trie", @@ -3112,9 +3123,9 @@ dependencies = [ [[package]] name = "pest_derive" -version = "2.8.7" +version = "2.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b4254325ecad416ab689e27ba51da03ba01a9632bc6e108f5fe7c3c4ad29d58" +checksum = "9e2dd6fc3b26b3462ee188aac870f5a41d398f1cd5e2408d16531bd71c9591fd" dependencies = [ "pest", "pest_generator", @@ -3122,22 +3133,22 @@ dependencies = [ [[package]] name = "pest_generator" -version = "2.8.7" +version = "2.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c4c0e91ead7a8f7acecbca6f003fc2e8282b1dbe2dd9c9d2f16aba42995e0a7" +checksum = "6a7a9205cfb6f596a9e8b689c0a15f9ceb7a1aafae7aaf788150ac65b29975b6" dependencies = [ "pest", "pest_meta", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "pest_meta" -version = "2.8.7" +version = "2.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9744bc48116fee06334924bb5f2bad41eed5e89bd26e29b0b799f9a3f82c210" +checksum = "85abd351c0de1e8384fc791a0737111a350394937e92b956b743dac12429f57c" dependencies = [ "pest", ] @@ -3232,7 +3243,7 @@ dependencies = [ "phf_shared 0.13.1", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3264,7 +3275,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "17.0.0" +version = "17.0.8" dependencies = [ "anyhow", "ast-grep-core", @@ -3333,7 +3344,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "17.0.0" +version = "17.0.8" dependencies = [ "async-trait", "libc", @@ -3345,7 +3356,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "17.0.0" +version = "17.0.8" dependencies = [ "anyhow", "arboard", @@ -3382,7 +3393,6 @@ dependencies = [ "regex", "serde", "serde_json", - "similar 3.1.1", "smallvec", "syntect", "tiktoken-rs", @@ -3398,7 +3408,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "17.0.0" +version = "17.0.8" dependencies = [ "anyhow", "brush-builtins", @@ -3482,7 +3492,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "17.0.0" +version = "17.0.8" dependencies = [ "dashmap", "globset", @@ -3551,7 +3561,7 @@ version = "0.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "crc32fast", "fdeflate", "flate2", @@ -3560,9 +3570,9 @@ dependencies = [ [[package]] name = "portable-atomic" -version = "1.13.1" +version = "1.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c33a9471896f1c69cecef8d20cbe2f7accd12527ce60845ff44c153bb2a21b49" +checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" [[package]] name = "portable-atomic-util" @@ -3618,7 +3628,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" dependencies = [ "proc-macro2", - "syn", + "syn 2.0.119", ] [[package]] @@ -3632,9 +3642,9 @@ dependencies = [ [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] @@ -3645,7 +3655,7 @@ version = "0.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "25485360a54d6861439d60facef26de713b1e126bf015ec8f98239467a2b82f7" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "chrono", "flate2", "procfs-core", @@ -3658,7 +3668,7 @@ version = "0.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6401bf7b6af22f78b563665d15a22e9aef27775b79b149a66ca022468a4e405" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "chrono", "hex", ] @@ -3695,15 +3705,6 @@ version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a993555f31e5a609f617c12db6250dedcac1b0a85076912c436e6fc9b2c8e6a3" -[[package]] -name = "quick-xml" -version = "0.39.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdcc8dd4e2f670d309a5f0e83fe36dfdc05af317008fea29144da1a2ac858e5e" -dependencies = [ - "memchr", -] - [[package]] name = "quick-xml" version = "0.41.0" @@ -3715,9 +3716,9 @@ dependencies = [ [[package]] name = "quote" -version = "1.0.46" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfbc457d0c7a0759a614551b11a6409e5951f6c7537be1f1b7682b9ae9230368" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -3822,34 +3823,34 @@ version = "0.5.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", ] [[package]] name = "ref-cast" -version = "1.0.25" +version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f354300ae66f76f1c85c5f84693f0ce81d747e2c3f21a45fef496d89c960bf7d" +checksum = "216e8f773d7923bcba9ceb86a86c93cabb3903a11872fc3f138c49630e50b96d" dependencies = [ "ref-cast-impl", ] [[package]] name = "ref-cast-impl" -version = "1.0.25" +version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7186006dcb21920990093f30e3dea63b7d6e977bf1256be20c3563a5db070da" +checksum = "2c9283685feec7d69af75fb0e858d5e7378f33fe4fc699383b2916ab9273e03c" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.3", ] [[package]] name = "regex" -version = "1.13.0" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2a0e75113e14dc5acb068cd0786884f214f1312650a3d36d269f5c4f3cdee8a2" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" dependencies = [ "aho-corasick", "memchr", @@ -3859,9 +3860,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.15" +version = "0.4.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f388202e4b80542a0921078cc23b6333bcf1409c1e3f86404cae4766a6131db" +checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" dependencies = [ "aho-corasick", "memchr", @@ -3939,7 +3940,7 @@ dependencies = [ "regex", "relative-path", "rustc_version", - "syn", + "syn 2.0.119", "unicode-ident", ] @@ -3970,7 +3971,7 @@ version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "errno", "libc", "linux-raw-sys", @@ -4009,9 +4010,9 @@ checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" [[package]] name = "self_cell" -version = "1.2.2" +version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b12e76d157a900eb52e81bc6e9f3069344290341720e9178cde2407113ac8d89" +checksum = "2ab42ca02749e120097e328d91d415325bdf43b1c72c4c8badf37375fe40a813" [[package]] name = "semver" @@ -4021,9 +4022,9 @@ checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -4031,29 +4032,29 @@ dependencies = [ [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.3", ] [[package]] name = "serde_json" -version = "1.0.150" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "indexmap", "itoa", @@ -4286,7 +4287,7 @@ dependencies = [ "heck", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4300,6 +4301,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "syn" +version = "3.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "synstructure" version = "0.13.2" @@ -4308,7 +4320,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4325,7 +4337,7 @@ dependencies = [ "regex-syntax", "serde", "serde_derive", - "thiserror 2.0.18", + "thiserror 2.0.19", "walkdir", "yaml-rust", ] @@ -4400,11 +4412,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.18" +version = "2.0.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" +checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" dependencies = [ - "thiserror-impl 2.0.18", + "thiserror-impl 2.0.19", ] [[package]] @@ -4415,18 +4427,18 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "thiserror-impl" -version = "2.0.18" +version = "2.0.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" +checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.3", ] [[package]] @@ -4480,9 +4492,9 @@ dependencies = [ [[package]] name = "tokio" -version = "1.52.3" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", @@ -4497,20 +4509,20 @@ dependencies = [ [[package]] name = "tokio-macros" -version = "2.7.0" +version = "2.7.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" +checksum = "6328af13490e73a9b4694030fafd93f8c8c6a9dede33e821c3fc63eddf8042ba" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", @@ -4518,6 +4530,7 @@ dependencies = [ "futures-sink", "futures-util", "hashbrown 0.15.5", + "libc", "pin-project-lite", "slab", "tokio", @@ -4593,7 +4606,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -5365,7 +5378,7 @@ dependencies = [ "clap", "memchr", "pi-uutils-ctx", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5452,7 +5465,7 @@ dependencies = [ "clap", "memchr", "pi-uutils-ctx", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5477,7 +5490,7 @@ dependencies = [ "parking_lot", "pi-uutils-ctx", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5492,7 +5505,7 @@ dependencies = [ "lscolors", "pi-uutils-ctx", "rustc-hash 2.1.3", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", "uutils_term_grid", ] @@ -5525,7 +5538,7 @@ dependencies = [ "pi-uutils-ctx", "rand 0.10.2", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5539,7 +5552,7 @@ dependencies = [ "libc", "pi-uutils-ctx", "rustc-hash 2.1.3", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", "windows-sys 0.61.2", ] @@ -5604,7 +5617,7 @@ dependencies = [ "indicatif", "libc", "pi-uutils-ctx", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", "windows-sys 0.61.2", ] @@ -5634,7 +5647,7 @@ dependencies = [ "num-traits", "parking_lot", "pi-uutils-ctx", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5706,7 +5719,7 @@ dependencies = [ "rayon", "self_cell", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5718,7 +5731,7 @@ dependencies = [ "parking_lot", "pi-uutils-ctx", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5733,7 +5746,7 @@ dependencies = [ "pi-uutils-ctx", "regex", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", ] @@ -5777,7 +5790,7 @@ dependencies = [ "pi-uutils-ctx", "rustix", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.19", "uucore 0.8.0", "windows-sys 0.61.2", ] @@ -5833,7 +5846,7 @@ dependencies = [ "libc", "pi-uutils-ctx", "rustix", - "thiserror 2.0.18", + "thiserror 2.0.19", "unicode-width 0.2.2", "uucore 0.8.0", ] @@ -5937,7 +5950,7 @@ dependencies = [ "sha2", "sha3", "sm3", - "thiserror 2.0.18", + "thiserror 2.0.19", "unic-langid", "unit-prefix", "uucore_procs 0.8.0", @@ -5962,7 +5975,7 @@ dependencies = [ "os_display", "rustc-hash 2.1.3", "rustix", - "thiserror 2.0.18", + "thiserror 2.0.19", "unic-langid", "uucore_procs 0.9.0", "wild", @@ -6007,9 +6020,9 @@ checksum = "0bb6d972f580f8223cb7052d8580aea2b7061e368cf476de32ea9457b19459ed" [[package]] name = "uuid" -version = "1.23.5" +version = "1.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ea5fab0d6c3c01ae70085a09cb03d4c7a1d6314e2b3e075392783396d724ca0a" +checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" dependencies = [ "js-sys", "wasm-bindgen", @@ -6121,7 +6134,7 @@ dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn", + "syn 2.0.119", "wasm-bindgen-shared", ] @@ -6136,9 +6149,9 @@ dependencies = [ [[package]] name = "wayland-backend" -version = "0.3.15" +version = "0.3.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2857dd20b54e916ec7253b3d6b4d5c4d7d4ca2c33c2e11c6c76a99bd8744755d" +checksum = "016ccf01d1c58b6f8999612813e17c9b2390f7d70671428869913310f83f54b8" dependencies = [ "cc", "downcast-rs", @@ -6149,11 +6162,11 @@ dependencies = [ [[package]] name = "wayland-client" -version = "0.31.14" +version = "0.31.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "645c7c96bb74690c3189b5c9cb4ca1627062bb23693a4fad9d8c3de958260144" +checksum = "e3c36a0f861ad76d0901f2800b46321410d9f73f2ea88aac0650d86c32688073" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "rustix", "wayland-backend", "wayland-scanner", @@ -6165,7 +6178,7 @@ version = "0.32.13" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "23d0c813de3daa2ed6520af85a3bd49b0e722a3078506899aa9686fea58dc4b6" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "wayland-backend", "wayland-client", "wayland-scanner", @@ -6177,7 +6190,7 @@ version = "0.3.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "eb04e52f7836d7c7976c78ca0250d61e33873c34156a2a1fc9474828ec268234" dependencies = [ - "bitflags 2.13.0", + "bitflags 2.13.1", "wayland-backend", "wayland-client", "wayland-protocols", @@ -6186,12 +6199,12 @@ dependencies = [ [[package]] name = "wayland-scanner" -version = "0.31.10" +version = "0.31.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c324a910fd86ebdc364a3e61ec1f11737d3b1d6c273c0239ee8ff4bc0d24b4a" +checksum = "338e30461b3a2b67d70eb30a6d89f8e0c93a833e07d2ae89085cd070c4a00ac0" dependencies = [ "proc-macro2", - "quick-xml 0.39.4", + "quick-xml", "quote", ] @@ -6345,7 +6358,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -6356,7 +6369,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -6649,7 +6662,7 @@ dependencies = [ "log", "os_pipe", "rustix", - "thiserror 2.0.18", + "thiserror 2.0.19", "tree_magic_mini", "wayland-backend", "wayland-client", @@ -6710,9 +6723,9 @@ dependencies = [ [[package]] name = "xxhash-rust" -version = "0.8.17" +version = "0.8.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "985eec839aaf2a1270af8f4ebcf63cf9401cfd90f0902f97c28d9f104ffbde72" +checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6" [[package]] name = "yaml-rust" @@ -6748,7 +6761,7 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "synstructure", ] @@ -6760,22 +6773,22 @@ checksum = "c6e61e59a957b7ccee15d2049f86e8bfd6f66968fcd88f018950662d9b86e675" [[package]] name = "zerocopy" -version = "0.8.54" +version = "0.8.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7cbbc0a705a0fd05cc3676525980d2bf5a9bc4adac6d6475209a7887cf59d19" +checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.54" +version = "0.8.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e2e817b7b52d0c7358d3246da9d69935ebb18116b2b102b4230dac079b4862f5" +checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -6795,7 +6808,7 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "synstructure", ] @@ -6831,7 +6844,7 @@ checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index 67fbffe89..1467fad4e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "17.0.0" +version = "17.0.8" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/README.md b/README.md index db1309caa..dde0fea97 100644 --- a/README.md +++ b/README.md @@ -26,6 +26,12 @@ The most capable agent surface that ships. Continuously tuned by real-world use **40+** providers · **32** built-in tools · **14** lsp ops · **28** dap ops · **~55k** lines of Rust core. +> [!NOTE] +> Pull requests are **temporarily open to everyone** as a trial. We previously +> required a vouch before accepting PRs; that requirement is lifted for now +> while we evaluate how open contributions go. Depending on the results, the +> vouch system may return. + ## Install **macOS · Linux** @@ -274,6 +280,16 @@ Setting-gated, off by default: `github`, `inspect_image`, `tts`, `checkpoint`, ` [Full reference →](https://omp.sh/docs/tools) +### Prompt controls + +Three standalone, lowercase words opt a turn into specialized agent behavior: + +- `ultrathink` — request careful multi-step reasoning and the highest supported automatic thinking effort. +- `orchestrate` — run substantial independent work through parallel subagents and verify each phase. +- `workflowz` — build a deterministic multi-subagent workflow with the active `task` tool. + +They trigger only in prose, not inside code spans, fenced code blocks, XML/HTML sections, identifiers, or paths. See [Magic keywords](docs/magic-keywords.md) for exact matching rules and configuration. + ## Forty-plus providers, hundreds of models, _one /model away_. Roles route work by intent. `default` for normal turns. `smol` for cheap subagent fan-out. `slow` for deep reasoning. `plan` for plan mode. `commit` for changelogs. Override at launch with `--smol`, `--slow`, or `--plan`; cycle through the configured models for the active role with `Ctrl+P`. Swap the active model mid-session with the `/model` slash command. @@ -298,6 +314,32 @@ OpenAI-compatible `/v1/models`. Local instances skip the key. Ollama `local` · Ollama Cloud · LM Studio `local` · llama.cpp `local` · vLLM `local` · LiteLLM +### Custom OpenAI-compatible providers + +Define custom providers in `~/.omp/agent/models.yml`: + +```yaml +providers: + spark: + baseUrl: http://192.168.10.223:8000/v1 + api: openai-completions + apiKey: dummy + models: + - id: minimax-m3 + name: MiniMax M3 + contextWindow: 100000 + maxTokens: 32000 +``` + +Run `omp models spark` to verify discovery. Then run `omp setup` and choose the model in the default-model step, or open `/model` in a session and assign it to the `default` role. + +To preconfigure the default without the picker, add the selector to `~/.omp/agent/config.yml`: + +```yaml +modelRoles: + default: spark/minimax-m3 +``` + ### Four knobs that make routing useful - **Custom providers** — Declare anything that speaks `openai-completions`, `openai-responses`, `openai-codex-responses`, `azure-openai-responses`, `anthropic-messages`, `google-generative-ai`, or `google-vertex` in `~/.omp/agent/models.yml`. @@ -556,12 +598,10 @@ For architecture and contribution guidelines, see [packages/coding-agent/DEVELOP ## Contributing -Issues are open to everyone. **Pull requests require a vouch** — PRs from -unvouched or denounced authors are closed automatically. If you're not yet -vouched, open a [Discussion](https://github.com/can1357/oh-my-pi/discussions) -and ask a maintainer to `!vouch` you rather than opening a PR (which would be -closed on sight). See **[CONTRIBUTING.md](CONTRIBUTING.md)** and -[`.github/VOUCHED.td`](.github/VOUCHED.td) for the full policy. +Issues and pull requests are open to everyone. Open PRs are currently a +**trial** — the previous vouch requirement is lifted while we evaluate how it +goes, and it may return. See **[CONTRIBUTING.md](CONTRIBUTING.md)** for +guidelines on contributing. --- diff --git a/bun.lock b/bun.lock index 6b8adec8b..587c7d05e 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,13 +69,14 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "17.0.0", + "version": "17.0.8", "bin": { "omp": "src/cli.ts", }, "dependencies": { "@agentclientprotocol/sdk": "catalog:", "@babel/parser": "catalog:", + "@babel/traverse": "catalog:", "@mozilla/readability": "catalog:", "@oh-my-pi/hashline": "catalog:", "@oh-my-pi/omp-stats": "catalog:", @@ -89,9 +90,14 @@ "@oh-my-pi/pi-wire": "catalog:", "@oh-my-pi/snapcompact": "catalog:", "@opentelemetry/api": "catalog:", + "@opentelemetry/api-logs": "catalog:", "@opentelemetry/context-async-hooks": "catalog:", + "@opentelemetry/exporter-logs-otlp-proto": "catalog:", + "@opentelemetry/exporter-metrics-otlp-proto": "catalog:", "@opentelemetry/exporter-trace-otlp-proto": "catalog:", "@opentelemetry/resources": "catalog:", + "@opentelemetry/sdk-logs": "catalog:", + "@opentelemetry/sdk-metrics": "catalog:", "@opentelemetry/sdk-trace-base": "catalog:", "@opentelemetry/sdk-trace-node": "catalog:", "@puppeteer/browsers": "catalog:", @@ -99,7 +105,6 @@ "@xterm/headless": "catalog:", "arktype": "catalog:", "chalk": "catalog:", - "diff": "catalog:", "fast-xml-parser": "catalog:", "handlebars": "catalog:", "header-generator": "catalog:", @@ -139,9 +144,9 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { - "diff": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "lru-cache": "catalog:", }, "devDependencies": { @@ -160,6 +165,7 @@ "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", "@oh-my-pi/pi-coding-agent": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "@oh-my-pi/typescript-edit-benchmark": "workspace:*", "clsx": "^2.1.1", @@ -182,13 +188,14 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "17.0.0", + "version": "17.0.8", "bin": { "mnemopi": "src/cli.ts", }, "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "lru-cache": "catalog:", }, @@ -208,7 +215,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "17.0.0", + "version": "17.0.8", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -216,7 +223,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -229,7 +236,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "17.0.0", + "version": "17.0.8", "bin": { "omp-stats": "./src/index.ts", }, @@ -250,12 +257,13 @@ "@types/bun": "catalog:", "@types/react": "catalog:", "@types/react-dom": "catalog:", + "linkedom": "catalog:", "postcss": "catalog:", }, }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "17.0.0", + "version": "17.0.8", "bin": { "omp-swarm": "src/cli.ts", }, @@ -271,7 +279,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -295,6 +303,7 @@ "@oh-my-pi/pi-agent-core": "catalog:", "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-coding-agent": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-tui": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "diff": "catalog:", @@ -309,7 +318,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "17.0.0", + "version": "17.0.8", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -322,7 +331,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "17.0.0", + "version": "17.0.8", "devDependencies": { "@types/bun": "catalog:", }, @@ -363,24 +372,29 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.2", - "@oh-my-pi/hashline": "17.0.0", - "@oh-my-pi/omp-stats": "17.0.0", - "@oh-my-pi/pi-agent-core": "17.0.0", - "@oh-my-pi/pi-ai": "17.0.0", - "@oh-my-pi/pi-catalog": "17.0.0", - "@oh-my-pi/pi-coding-agent": "17.0.0", - "@oh-my-pi/pi-mnemopi": "17.0.0", - "@oh-my-pi/pi-natives": "17.0.0", - "@oh-my-pi/pi-tui": "17.0.0", - "@oh-my-pi/pi-utils": "17.0.0", - "@oh-my-pi/pi-wire": "17.0.0", - "@oh-my-pi/snapcompact": "17.0.0", + "@oh-my-pi/hashline": "17.0.8", + "@oh-my-pi/omp-stats": "17.0.8", + "@oh-my-pi/pi-agent-core": "17.0.8", + "@oh-my-pi/pi-ai": "17.0.8", + "@oh-my-pi/pi-catalog": "17.0.8", + "@oh-my-pi/pi-coding-agent": "17.0.8", + "@oh-my-pi/pi-mnemopi": "17.0.8", + "@oh-my-pi/pi-natives": "17.0.8", + "@oh-my-pi/pi-tui": "17.0.8", + "@oh-my-pi/pi-utils": "17.0.8", + "@oh-my-pi/pi-wire": "17.0.8", + "@oh-my-pi/snapcompact": "17.0.8", "@opentelemetry/api": "^1.9.1", - "@opentelemetry/context-async-hooks": "^2.7.1", + "@opentelemetry/api-logs": "^0.220.0", + "@opentelemetry/context-async-hooks": "^2.9.0", + "@opentelemetry/exporter-logs-otlp-proto": "^0.220.0", + "@opentelemetry/exporter-metrics-otlp-proto": "^0.220.0", "@opentelemetry/exporter-trace-otlp-proto": "^0.220.0", - "@opentelemetry/resources": "^2.7.1", - "@opentelemetry/sdk-trace-base": "^2.7.1", - "@opentelemetry/sdk-trace-node": "^2.7.1", + "@opentelemetry/resources": "^2.9.0", + "@opentelemetry/sdk-logs": "^0.220.0", + "@opentelemetry/sdk-metrics": "^2.9.0", + "@opentelemetry/sdk-trace-base": "^2.9.0", + "@opentelemetry/sdk-trace-node": "^2.9.0", "@puppeteer/browsers": "^3.0.6", "@tailwindcss/node": "^4.3.2", "@tailwindcss/vite": "^4.3.2", @@ -481,23 +495,23 @@ "@babel/types": ["@babel/types@7.29.7", "", { "dependencies": { "@babel/helper-string-parser": "^7.29.7", "@babel/helper-validator-identifier": "^7.29.7" } }, "sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA=="], - "@biomejs/biome": ["@biomejs/biome@2.5.3", "", { "optionalDependencies": { "@biomejs/cli-darwin-arm64": "2.5.3", "@biomejs/cli-darwin-x64": "2.5.3", "@biomejs/cli-linux-arm64": "2.5.3", "@biomejs/cli-linux-arm64-musl": "2.5.3", "@biomejs/cli-linux-x64": "2.5.3", "@biomejs/cli-linux-x64-musl": "2.5.3", "@biomejs/cli-win32-arm64": "2.5.3", "@biomejs/cli-win32-x64": "2.5.3" }, "bin": { "biome": "bin/biome" } }, "sha512-MrJswFdei9EfDwwUy2tQrPDpK0AO+RmMFvBoaaJ6ayBc3sUbHdCE+XG5N8vp+5So41ZupZJQm0roHFFhMGVD7A=="], + "@biomejs/biome": ["@biomejs/biome@2.5.4", "", { "optionalDependencies": { "@biomejs/cli-darwin-arm64": "2.5.4", "@biomejs/cli-darwin-x64": "2.5.4", "@biomejs/cli-linux-arm64": "2.5.4", "@biomejs/cli-linux-arm64-musl": "2.5.4", "@biomejs/cli-linux-x64": "2.5.4", "@biomejs/cli-linux-x64-musl": "2.5.4", "@biomejs/cli-win32-arm64": "2.5.4", "@biomejs/cli-win32-x64": "2.5.4" }, "bin": { "biome": "bin/biome" } }, "sha512-xy5FNE5kQJKyK5MR1gJy6ztXYx4WBAbYGlK04lMEgmyPRWKybY9NFwiG9yo0XdzOU8Xvhj41u034J1ywfoWfMw=="], - "@biomejs/cli-darwin-arm64": ["@biomejs/cli-darwin-arm64@2.5.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-QhYP9muVQ0nUO5zztFuPbEwi4+94sJWVjaZds9aMi1l/KNZBiUjdiSUrGHsTaMGDXrYl+r4AS2sUKfgH3w+V3g=="], + "@biomejs/cli-darwin-arm64": ["@biomejs/cli-darwin-arm64@2.5.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-4o3NFRobXHynkgcFVrlZsoDAFtF2ldlEGN8sORSws5ZQqyY4PXnPUIylu4ksfyHuwkfvDREuWh3JK+niRwGq3w=="], - "@biomejs/cli-darwin-x64": ["@biomejs/cli-darwin-x64@2.5.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-NC1Ss13UaW7QZX+y8j44bF7AP0jSJdBl6iRhe0MAkvaSqZy+mWg3GaXsrb+eSoHoGDBtaXWEbMVV0iVN2cZ7cQ=="], + "@biomejs/cli-darwin-x64": ["@biomejs/cli-darwin-x64@2.5.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-D32P5HkU2Y6PySuC/WsVDTOgsDwVFmujzhhhOQjajtATpVWFDXuVd3oRbsWNSEA+aaFzyzZm22szsyydBYlSyQ=="], - "@biomejs/cli-linux-arm64": ["@biomejs/cli-linux-arm64@2.5.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-ksx1KWeyYW18ILL04msF/J4ZBtBDN33znYK8Z/aNv/vlBVxL9/g3mGP+omgHJKy4+KWbK87vcmmpmurfNjSgiA=="], + "@biomejs/cli-linux-arm64": ["@biomejs/cli-linux-arm64@2.5.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-pSEfW7B8kTsXUjUxC1xVVK+y85Ht3C5XxZ9gclmC7/3Ku9Vqz8jmI7k0p/BNIjQ6t4sFERI2sFeH73ybiZl6YQ=="], - "@biomejs/cli-linux-arm64-musl": ["@biomejs/cli-linux-arm64-musl@2.5.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-fccix0w6xp6csCXgxeC0dU/3ecgRQal0y+cv2SP9ajNlhe7Yrk2Ug7UDe2j9AT9ZDYitkXpvUKgZjjuoYeP4Vg=="], + "@biomejs/cli-linux-arm64-musl": ["@biomejs/cli-linux-arm64-musl@2.5.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-Rpm5/AT1m+DlJmUoYvS4/vXc+0tXJPJ2NQz25TGPyHVF5JrWy75PE0GH6kVxsKtQDuCH4OgzquZq0R4kj/wCVg=="], - "@biomejs/cli-linux-x64": ["@biomejs/cli-linux-x64@2.5.3", "", { "os": "linux", "cpu": "x64" }, "sha512-yMkJtilsgvILDcVkh187aVLTb64xYsrxYajx5kym+r1ULkO5HUOfu9AYKLGQbOVLwJtT2utNw7hhFNg+17mUYA=="], + "@biomejs/cli-linux-x64": ["@biomejs/cli-linux-x64@2.5.4", "", { "os": "linux", "cpu": "x64" }, "sha512-FNxojWJkL7EajAuzBgoLe0T2G0y112M4lBrDIFl/DomFTx8yqenYOIdsRLNXvOvBBofE8hJi85LjzLmBDpY7/Q=="], - "@biomejs/cli-linux-x64-musl": ["@biomejs/cli-linux-x64-musl@2.5.3", "", { "os": "linux", "cpu": "x64" }, "sha512-O/yU9YKRUiHhmcjF2f38PSjseVk3G4VLWYc0G2HWpzdBVREV6G8IGWIVEFf7MFPfWIzNUIvPsEjeAZQIOgnLcQ=="], + "@biomejs/cli-linux-x64-musl": ["@biomejs/cli-linux-x64-musl@2.5.4", "", { "os": "linux", "cpu": "x64" }, "sha512-aby/PohmmgbShcHqFsZVzG8H6D98+P+A6xRWRrQcLW1pCjabcov5UUlke4UqNQBYTkDQav+jB4zyyDDeKB2GaA=="], - "@biomejs/cli-win32-arm64": ["@biomejs/cli-win32-arm64@2.5.3", "", { "os": "win32", "cpu": "arm64" }, "sha512-cX5z+GYwRcqEok0AH3KSfQGgqYd0Nomfp6Fbe1uiTtELE38hdH2k842wQ9wLNaF/JJ7r4rjJQ4VR+ce+fRmQbw=="], + "@biomejs/cli-win32-arm64": ["@biomejs/cli-win32-arm64@2.5.4", "", { "os": "win32", "cpu": "arm64" }, "sha512-emoXexPZIPAZkz2RKmA95WJUqK3I5MJNYtwEbL5ESciRzhmFMMyekDhNG8hpeOaK+ZGRDxAU4wvGuA5IHQ0h0w=="], - "@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.3", "", { "os": "win32", "cpu": "x64" }, "sha512-ExSaJWi4/u6+GXCszlSKpWSjKNbDseAYqqkCznsCsZ/4uidZ/BEqsCc5/3ctlq6dfIubdIIRSVLC/PG9xPl70Q=="], + "@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.4", "", { "os": "win32", "cpu": "x64" }, "sha512-U1jaluLw1qQc2Tx7/CeSoL9N5XcqIH+GWjpUAy1ouB5nVjSCMNO+NNHdY3RAs8zxNurLWAdj6pehQdCA2zyU+Q=="], "@bufbuild/protobuf": ["@bufbuild/protobuf@2.12.1", "", {}, "sha512-BvAMfS6LrgZiryOAZ4pBYucu4wG/Ei/9o9DZ9akbREnMLbPJiom2i8b9C8IsKErQoiKqVhrerzt3kOT/RrzLHg=="], @@ -521,7 +535,7 @@ "@huggingface/jinja": ["@huggingface/jinja@0.5.9", "", {}, "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw=="], - "@huggingface/tasks": ["@huggingface/tasks@0.21.25", "", {}, "sha512-u71rRx80Ynzy2oEDpRTSgOLyrzAV86gtnt4ajGHHI+1SQmsqmIUyzZG4Wi0TavDKdRsTa5QC/REDeJ3cFvkxvA=="], + "@huggingface/tasks": ["@huggingface/tasks@0.21.26", "", {}, "sha512-6QGX2qBFEf3zU+qRz+kSTI8weoUpLtNINnl8i7NUC+VB1UBxKvgPrXGCLEEkUwcRGqC0DBpJBwB0OPFg1Aa25g=="], "@huggingface/tokenizers": ["@huggingface/tokenizers@0.1.3", "", {}, "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA=="], @@ -633,75 +647,75 @@ "@napi-rs/cross-toolchain": ["@napi-rs/cross-toolchain@1.0.3", "", { "dependencies": { "@napi-rs/lzma": "^1.4.5", "@napi-rs/tar": "^1.1.0", "debug": "^4.4.1" }, "peerDependencies": { "@napi-rs/cross-toolchain-arm64-target-aarch64": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-armv7": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-ppc64le": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-s390x": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-x86_64": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-aarch64": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-armv7": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-ppc64le": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-s390x": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-x86_64": "^1.0.3" }, "optionalPeers": ["@napi-rs/cross-toolchain-arm64-target-aarch64", "@napi-rs/cross-toolchain-arm64-target-armv7", "@napi-rs/cross-toolchain-arm64-target-ppc64le", "@napi-rs/cross-toolchain-arm64-target-s390x", "@napi-rs/cross-toolchain-arm64-target-x86_64", "@napi-rs/cross-toolchain-x64-target-aarch64", "@napi-rs/cross-toolchain-x64-target-armv7", "@napi-rs/cross-toolchain-x64-target-ppc64le", "@napi-rs/cross-toolchain-x64-target-s390x", "@napi-rs/cross-toolchain-x64-target-x86_64"] }, "sha512-ENPfLe4937bsKVTDA6zdABx4pq9w0tHqRrJHyaGxgaPq03a2Bd1unD5XSKjXJjebsABJ+MjAv1A2OvCgK9yehg=="], - "@napi-rs/lzma": ["@napi-rs/lzma@1.4.5", "", { "optionalDependencies": { "@napi-rs/lzma-android-arm-eabi": "1.4.5", "@napi-rs/lzma-android-arm64": "1.4.5", "@napi-rs/lzma-darwin-arm64": "1.4.5", "@napi-rs/lzma-darwin-x64": "1.4.5", "@napi-rs/lzma-freebsd-x64": "1.4.5", "@napi-rs/lzma-linux-arm-gnueabihf": "1.4.5", "@napi-rs/lzma-linux-arm64-gnu": "1.4.5", "@napi-rs/lzma-linux-arm64-musl": "1.4.5", "@napi-rs/lzma-linux-ppc64-gnu": "1.4.5", "@napi-rs/lzma-linux-riscv64-gnu": "1.4.5", "@napi-rs/lzma-linux-s390x-gnu": "1.4.5", "@napi-rs/lzma-linux-x64-gnu": "1.4.5", "@napi-rs/lzma-linux-x64-musl": "1.4.5", "@napi-rs/lzma-wasm32-wasi": "1.4.5", "@napi-rs/lzma-win32-arm64-msvc": "1.4.5", "@napi-rs/lzma-win32-ia32-msvc": "1.4.5", "@napi-rs/lzma-win32-x64-msvc": "1.4.5" } }, "sha512-zS5LuN1OBPAyZpda2ZZgYOEDC+xecUdAGnrvbYzjnLXkrq/OBC3B9qcRvlxbDR3k5H/gVfvef1/jyUqPknqjbg=="], + "@napi-rs/lzma": ["@napi-rs/lzma@1.5.1", "", { "optionalDependencies": { "@napi-rs/lzma-android-arm-eabi": "1.5.1", "@napi-rs/lzma-android-arm64": "1.5.1", "@napi-rs/lzma-darwin-arm64": "1.5.1", "@napi-rs/lzma-darwin-x64": "1.5.1", "@napi-rs/lzma-freebsd-x64": "1.5.1", "@napi-rs/lzma-linux-arm-gnueabihf": "1.5.1", "@napi-rs/lzma-linux-arm64-gnu": "1.5.1", "@napi-rs/lzma-linux-arm64-musl": "1.5.1", "@napi-rs/lzma-linux-ppc64-gnu": "1.5.1", "@napi-rs/lzma-linux-riscv64-gnu": "1.5.1", "@napi-rs/lzma-linux-s390x-gnu": "1.5.1", "@napi-rs/lzma-linux-x64-gnu": "1.5.1", "@napi-rs/lzma-linux-x64-musl": "1.5.1", "@napi-rs/lzma-wasm32-wasi": "1.5.1", "@napi-rs/lzma-win32-arm64-msvc": "1.5.1", "@napi-rs/lzma-win32-ia32-msvc": "1.5.1", "@napi-rs/lzma-win32-x64-msvc": "1.5.1" } }, "sha512-sgOZ89+y8cDbY+3WbzR8CtIhCuFRWotZ9/2PjPVDJHz6np5KFTAev0DrwiyTJTgFsCRDhfGlbmhMgyhHbWdZ6g=="], - "@napi-rs/lzma-android-arm-eabi": ["@napi-rs/lzma-android-arm-eabi@1.4.5", "", { "os": "android", "cpu": "arm" }, "sha512-Up4gpyw2SacmyKWWEib06GhiDdF+H+CCU0LAV8pnM4aJIDqKKd5LHSlBht83Jut6frkB0vwEPmAkv4NjQ5u//Q=="], + "@napi-rs/lzma-android-arm-eabi": ["@napi-rs/lzma-android-arm-eabi@1.5.1", "", { "os": "android", "cpu": "arm" }, "sha512-sahBe4ko2Z69NPTddaX6ZgbQZu9SDoITxw1S3dWl1gAGynZG34qHHCT8UaUMFxf3h3zMhCJjEzz4basaBxiTuQ=="], - "@napi-rs/lzma-android-arm64": ["@napi-rs/lzma-android-arm64@1.4.5", "", { "os": "android", "cpu": "arm64" }, "sha512-uwa8sLlWEzkAM0MWyoZJg0JTD3BkPknvejAFG2acUA1raXM8jLrqujWCdOStisXhqQjZ2nDMp3FV6cs//zjfuQ=="], + "@napi-rs/lzma-android-arm64": ["@napi-rs/lzma-android-arm64@1.5.1", "", { "os": "android", "cpu": "arm64" }, "sha512-7tkQAJJuBHxAxiEBNFgSTpvrtGpbwZYYJUSOmGEK3OfbdbNeoT2rdBxpM/gY1s+itEVbtOSlpaRPPG19MnwOzA=="], - "@napi-rs/lzma-darwin-arm64": ["@napi-rs/lzma-darwin-arm64@1.4.5", "", { "os": "darwin", "cpu": "arm64" }, "sha512-0Y0TQLQ2xAjVabrMDem1NhIssOZzF/y/dqetc6OT8mD3xMTDtF8u5BqZoX3MyPc9FzpsZw4ksol+w7DsxHrpMA=="], + "@napi-rs/lzma-darwin-arm64": ["@napi-rs/lzma-darwin-arm64@1.5.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-XWX8gtF+GHGk3nH3Wm3QUZNcxw9QHsFVZz3MzVLhWWHhceede1J4/vD+3dj3E1iKB9G6mualaZxOoD08R3E+7g=="], - "@napi-rs/lzma-darwin-x64": ["@napi-rs/lzma-darwin-x64@1.4.5", "", { "os": "darwin", "cpu": "x64" }, "sha512-vR2IUyJY3En+V1wJkwmbGWcYiT8pHloTAWdW4pG24+51GIq+intst6Uf6D/r46citObGZrlX0QvMarOkQeHWpw=="], + "@napi-rs/lzma-darwin-x64": ["@napi-rs/lzma-darwin-x64@1.5.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-CfsqUpMTI1z8enrA/b+GcHM6YDI8D0kqCiqPYEnst4rbOABQ9KZ92ybTTNnlnZ7A017WoMZKUEWc36KXDwi0xg=="], - "@napi-rs/lzma-freebsd-x64": ["@napi-rs/lzma-freebsd-x64@1.4.5", "", { "os": "freebsd", "cpu": "x64" }, "sha512-XpnYQC5SVovO35tF0xGkbHYjsS6kqyNCjuaLQ2dbEblFRr5cAZVvsJ/9h7zj/5FluJPJRDojVNxGyRhTp4z2lw=="], + "@napi-rs/lzma-freebsd-x64": ["@napi-rs/lzma-freebsd-x64@1.5.1", "", { "os": "freebsd", "cpu": "x64" }, "sha512-bTyNfg90FXIgE61U7l14aMmVOqRQ6AyP5JMT3jmCStaZI18apLNPdzZ8i7yqxZfKvRMVfPjE2brXIw27c+RRgA=="], - "@napi-rs/lzma-linux-arm-gnueabihf": ["@napi-rs/lzma-linux-arm-gnueabihf@1.4.5", "", { "os": "linux", "cpu": "arm" }, "sha512-ic1ZZMoRfRMwtSwxkyw4zIlbDZGC6davC9r+2oX6x9QiF247BRqqT94qGeL5ZP4Vtz0Hyy7TEViWhx5j6Bpzvw=="], + "@napi-rs/lzma-linux-arm-gnueabihf": ["@napi-rs/lzma-linux-arm-gnueabihf@1.5.1", "", { "os": "linux", "cpu": "arm" }, "sha512-vNE+D8nrw+eOkBsdKCsmDhowDV3pIMKXEhedvXfbgrWbrO7GlZJH+RXL+X+RYLxGwi8Ym61ZMt15sIOnNmh9Sw=="], - "@napi-rs/lzma-linux-arm64-gnu": ["@napi-rs/lzma-linux-arm64-gnu@1.4.5", "", { "os": "linux", "cpu": "arm64" }, "sha512-asEp7FPd7C1Yi6DQb45a3KPHKOFBSfGuJWXcAd4/bL2Fjetb2n/KK2z14yfW8YC/Fv6x3rBM0VAZKmJuz4tysg=="], + "@napi-rs/lzma-linux-arm64-gnu": ["@napi-rs/lzma-linux-arm64-gnu@1.5.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-csUem4WgoKGTprv/pOPm9UIWbb+hrfUwYXefpTHPAEGVFLl5behEFabisJ7FtihCa3yG2Efcl+yw25rlhhrIYw=="], - "@napi-rs/lzma-linux-arm64-musl": ["@napi-rs/lzma-linux-arm64-musl@1.4.5", "", { "os": "linux", "cpu": "arm64" }, "sha512-yWjcPDgJ2nIL3KNvi4536dlT/CcCWO0DUyEOlBs/SacG7BeD6IjGh6yYzd3/X1Y3JItCbZoDoLUH8iB1lTXo3w=="], + "@napi-rs/lzma-linux-arm64-musl": ["@napi-rs/lzma-linux-arm64-musl@1.5.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-kB/xhlVN1eLvVmDJSKZEjp5Gg2xDYexNrB5jwpSMbOkeGS6N9AasByPBg5VqCpMYC+zZi7DM458DRhtWYhqXTQ=="], - "@napi-rs/lzma-linux-ppc64-gnu": ["@napi-rs/lzma-linux-ppc64-gnu@1.4.5", "", { "os": "linux", "cpu": "ppc64" }, "sha512-0XRhKuIU/9ZjT4WDIG/qnX7Xz7mSQHYZo9Gb3MP2gcvBgr6BA4zywQ9k3gmQaPn9ECE+CZg2V7DV7kT+x2pUMQ=="], + "@napi-rs/lzma-linux-ppc64-gnu": ["@napi-rs/lzma-linux-ppc64-gnu@1.5.1", "", { "os": "linux", "cpu": "ppc64" }, "sha512-s28RW0W1yBWQc1nbPdF7tp14koqslY3ZWLVI8uaanX292Dc6ezd4NPVwxEoCNBVON/oD7BmUbWGtyFvmm7dQ5A=="], - "@napi-rs/lzma-linux-riscv64-gnu": ["@napi-rs/lzma-linux-riscv64-gnu@1.4.5", "", { "os": "linux", "cpu": "none" }, "sha512-QrqDIPEUUB23GCpyQj/QFyMlr8SGxxyExeZz9OWFnHfb70kXdTLWrHS/hEI1Ru+lSbQ/6xRqeoGyQ4Aqdg+/RA=="], + "@napi-rs/lzma-linux-riscv64-gnu": ["@napi-rs/lzma-linux-riscv64-gnu@1.5.1", "", { "os": "linux", "cpu": "none" }, "sha512-+lGNwYlIN14YPMTNvYtIJJqHFevDTd6Juw/1NmXbWx/iRd/LLrjhlM/yluMX6pxs6NkOGsuuEXJJrbbEUS59OQ=="], - "@napi-rs/lzma-linux-s390x-gnu": ["@napi-rs/lzma-linux-s390x-gnu@1.4.5", "", { "os": "linux", "cpu": "s390x" }, "sha512-k8RVM5aMhW86E9H0QXdquwojew4H3SwPxbRVbl49/COJQWCUjGi79X6mYruMnMPEznZinUiT1jgKbFo2A00NdA=="], + "@napi-rs/lzma-linux-s390x-gnu": ["@napi-rs/lzma-linux-s390x-gnu@1.5.1", "", { "os": "linux", "cpu": "s390x" }, "sha512-PB44FFWWFrLeQowhcep1hPD1YcLqKlnnY60RMU74qrxTlr4YGEyzeMItJqh2uivBfv9kQScOF/B0J9+Vab/oyw=="], - "@napi-rs/lzma-linux-x64-gnu": ["@napi-rs/lzma-linux-x64-gnu@1.4.5", "", { "os": "linux", "cpu": "x64" }, "sha512-6rMtBgnIq2Wcl1rQdZsnM+rtCcVCbws1nF8S2NzaUsVaZv8bjrPiAa0lwg4Eqnn1d9lgwqT+cZgm5m+//K08Kw=="], + "@napi-rs/lzma-linux-x64-gnu": ["@napi-rs/lzma-linux-x64-gnu@1.5.1", "", { "os": "linux", "cpu": "x64" }, "sha512-oTXEIha4SsuXdTA4Iyskj0kpdx2yVXdhd75c2v3xGrHFfVMsbhTPZU/nMPL4sWKo4pBHm3aucLaqGlF696dTyQ=="], - "@napi-rs/lzma-linux-x64-musl": ["@napi-rs/lzma-linux-x64-musl@1.4.5", "", { "os": "linux", "cpu": "x64" }, "sha512-eiadGBKi7Vd0bCArBUOO/qqRYPHt/VQVvGyYvDFt6C2ZSIjlD+HuOl+2oS1sjf4CFjK4eDIog6EdXnL0NE6iyQ=="], + "@napi-rs/lzma-linux-x64-musl": ["@napi-rs/lzma-linux-x64-musl@1.5.1", "", { "os": "linux", "cpu": "x64" }, "sha512-I3nsYrWtrW9JpeCr+mkJIVDt0HY3m6qVUBs5vTtoIvJQxwqf1PBXSy5IS7T53ksQFH2kd2UX8rLxJ7B4WISpZg=="], - "@napi-rs/lzma-wasm32-wasi": ["@napi-rs/lzma-wasm32-wasi@1.4.5", "", { "dependencies": { "@napi-rs/wasm-runtime": "^1.0.3" }, "cpu": "none" }, "sha512-+VyHHlr68dvey6fXc2hehw9gHVFIW3TtGF1XkcbAu65qVXsA9D/T+uuoRVqhE+JCyFHFrO0ixRbZDRK1XJt1sA=="], + "@napi-rs/lzma-wasm32-wasi": ["@napi-rs/lzma-wasm32-wasi@1.5.1", "", { "dependencies": { "@emnapi/core": "1.11.2", "@emnapi/runtime": "1.11.2", "@napi-rs/wasm-runtime": "^1.1.6" }, "cpu": "none" }, "sha512-gy3wwPBa6+XEyA4fUzq6CClrXA1ajXjuVf5zbnHytJRgoHznj+mvpU3+co2fxXwqTCmIpn6KrzqH5bRDztBPhA=="], - "@napi-rs/lzma-win32-arm64-msvc": ["@napi-rs/lzma-win32-arm64-msvc@1.4.5", "", { "os": "win32", "cpu": "arm64" }, "sha512-eewnqvIyyhHi3KaZtBOJXohLvwwN27gfS2G/YDWdfHlbz1jrmfeHAmzMsP5qv8vGB+T80TMHNkro4kYjeh6Deg=="], + "@napi-rs/lzma-win32-arm64-msvc": ["@napi-rs/lzma-win32-arm64-msvc@1.5.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-dK+huOsHiyH6oJjij+cnjqFCakk2HgWmpI12Xm4pLUyPphe4ebYoJBgehaNAxprmjFqBQ7nL95YPVz9BHyqmPg=="], - "@napi-rs/lzma-win32-ia32-msvc": ["@napi-rs/lzma-win32-ia32-msvc@1.4.5", "", { "os": "win32", "cpu": "ia32" }, "sha512-OeacFVRCJOKNU/a0ephUfYZ2Yt+NvaHze/4TgOwJ0J0P4P7X1mHzN+ig9Iyd74aQDXYqc7kaCXA2dpAOcH87Cg=="], + "@napi-rs/lzma-win32-ia32-msvc": ["@napi-rs/lzma-win32-ia32-msvc@1.5.1", "", { "os": "win32", "cpu": "ia32" }, "sha512-dGE8L+0EQ+GyU9ap9InqB/t/PmPG/bLj918q7OsJ29FuTdn8fK4OX3U4IQZhylHIA+/dQ/SXJk5n4yfah2XVvA=="], - "@napi-rs/lzma-win32-x64-msvc": ["@napi-rs/lzma-win32-x64-msvc@1.4.5", "", { "os": "win32", "cpu": "x64" }, "sha512-T4I1SamdSmtyZgDXGAGP+y5LEK5vxHUFwe8mz6D4R7Sa5/WCxTcCIgPJ9BD7RkpO17lzhlaM2vmVvMy96Lvk9Q=="], + "@napi-rs/lzma-win32-x64-msvc": ["@napi-rs/lzma-win32-x64-msvc@1.5.1", "", { "os": "win32", "cpu": "x64" }, "sha512-EKW4t/iqdCT/xnd5t9oXLvVER/PMNAWXKqUAl3fgvUcOILeZIIht77/dVnfFcc9htA/DCBXC/6YQWdW+LusjFA=="], - "@napi-rs/tar": ["@napi-rs/tar@1.1.0", "", { "optionalDependencies": { "@napi-rs/tar-android-arm-eabi": "1.1.0", "@napi-rs/tar-android-arm64": "1.1.0", "@napi-rs/tar-darwin-arm64": "1.1.0", "@napi-rs/tar-darwin-x64": "1.1.0", "@napi-rs/tar-freebsd-x64": "1.1.0", "@napi-rs/tar-linux-arm-gnueabihf": "1.1.0", "@napi-rs/tar-linux-arm64-gnu": "1.1.0", "@napi-rs/tar-linux-arm64-musl": "1.1.0", "@napi-rs/tar-linux-ppc64-gnu": "1.1.0", "@napi-rs/tar-linux-s390x-gnu": "1.1.0", "@napi-rs/tar-linux-x64-gnu": "1.1.0", "@napi-rs/tar-linux-x64-musl": "1.1.0", "@napi-rs/tar-wasm32-wasi": "1.1.0", "@napi-rs/tar-win32-arm64-msvc": "1.1.0", "@napi-rs/tar-win32-ia32-msvc": "1.1.0", "@napi-rs/tar-win32-x64-msvc": "1.1.0" } }, "sha512-7cmzIu+Vbupriudo7UudoMRH2OA3cTw67vva8MxeoAe5S7vPFI7z0vp0pMXiA25S8IUJefImQ90FeJjl8fjEaQ=="], + "@napi-rs/tar": ["@napi-rs/tar@1.1.1", "", { "optionalDependencies": { "@napi-rs/tar-android-arm-eabi": "1.1.1", "@napi-rs/tar-android-arm64": "1.1.1", "@napi-rs/tar-darwin-arm64": "1.1.1", "@napi-rs/tar-darwin-x64": "1.1.1", "@napi-rs/tar-freebsd-x64": "1.1.1", "@napi-rs/tar-linux-arm-gnueabihf": "1.1.1", "@napi-rs/tar-linux-arm64-gnu": "1.1.1", "@napi-rs/tar-linux-arm64-musl": "1.1.1", "@napi-rs/tar-linux-ppc64-gnu": "1.1.1", "@napi-rs/tar-linux-s390x-gnu": "1.1.1", "@napi-rs/tar-linux-x64-gnu": "1.1.1", "@napi-rs/tar-linux-x64-musl": "1.1.1", "@napi-rs/tar-wasm32-wasi": "1.1.1", "@napi-rs/tar-win32-arm64-msvc": "1.1.1", "@napi-rs/tar-win32-ia32-msvc": "1.1.1", "@napi-rs/tar-win32-x64-msvc": "1.1.1" } }, "sha512-p6q2HhUc5vwH1CNwfOcrhLoxfgn8ust8Sqlfx+sA4VzAcp1cMbvbkl99tZZlDqOjCHgQNSiTfk/yWPjl/D42qA=="], - "@napi-rs/tar-android-arm-eabi": ["@napi-rs/tar-android-arm-eabi@1.1.0", "", { "os": "android", "cpu": "arm" }, "sha512-h2Ryndraj/YiKgMV/r5by1cDusluYIRT0CaE0/PekQ4u+Wpy2iUVqvzVU98ZPnhXaNeYxEvVJHNGafpOfaD0TA=="], + "@napi-rs/tar-android-arm-eabi": ["@napi-rs/tar-android-arm-eabi@1.1.1", "", { "os": "android", "cpu": "arm" }, "sha512-cAhnA10cSusAUbcE9HtjQY/tZ9BH/0w2sKtRcQc94TzIlnm7QSr1htJSd/PPrbWNPtrv1orXb2CkrHlVlbnlHA=="], - "@napi-rs/tar-android-arm64": ["@napi-rs/tar-android-arm64@1.1.0", "", { "os": "android", "cpu": "arm64" }, "sha512-DJFyQHr1ZxNZorm/gzc1qBNLF/FcKzcH0V0Vwan5P+o0aE2keQIGEjJ09FudkF9v6uOuJjHCVDdK6S6uHtShAw=="], + "@napi-rs/tar-android-arm64": ["@napi-rs/tar-android-arm64@1.1.1", "", { "os": "android", "cpu": "arm64" }, "sha512-EslUWHCDBY/g5abTPBiHLsMaML4GagV0TXLm5WL9hAjx/DDtlxz9fegMb77RJ+f7nFLOIsUxF/3QWFvgOT0sMQ=="], - "@napi-rs/tar-darwin-arm64": ["@napi-rs/tar-darwin-arm64@1.1.0", "", { "os": "darwin", "cpu": "arm64" }, "sha512-Zz2sXRzjIX4e532zD6xm2SjXEym6MkvfCvL2RMpG2+UwNVDVscHNcz3d47Pf3sysP2e2af7fBB3TIoK2f6trPw=="], + "@napi-rs/tar-darwin-arm64": ["@napi-rs/tar-darwin-arm64@1.1.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-+A42/6ES5G9CQ35BOwzwA+WBjLID28r2jNPgc0dteD2hhClIhng0mva7D2ujUlXBNmgNOsr1LHn3stA4uTf4NQ=="], - "@napi-rs/tar-darwin-x64": ["@napi-rs/tar-darwin-x64@1.1.0", "", { "os": "darwin", "cpu": "x64" }, "sha512-EI+CptIMNweT0ms9S3mkP/q+J6FNZ1Q6pvpJOEcWglRfyfQpLqjlC0O+dptruTPE8VamKYuqdjxfqD8hifZDOA=="], + "@napi-rs/tar-darwin-x64": ["@napi-rs/tar-darwin-x64@1.1.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-RYtE8w1dkEvj8hSJCDV5Jw0Rz2i13fsM7u893zv5O9n/4Ad5GNsw/f4RQ7/0YGSFaenkVxqPFrjmEvUHlKzsrg=="], - "@napi-rs/tar-freebsd-x64": ["@napi-rs/tar-freebsd-x64@1.1.0", "", { "os": "freebsd", "cpu": "x64" }, "sha512-J0PIqX+pl6lBIAckL/c87gpodLbjZB1OtIK+RDscKC9NLdpVv6VGOxzUV/fYev/hctcE8EfkLbgFOfpmVQPg2g=="], + "@napi-rs/tar-freebsd-x64": ["@napi-rs/tar-freebsd-x64@1.1.1", "", { "os": "freebsd", "cpu": "x64" }, "sha512-rEepBvCJUwcuvUYkY83e8aot8RsR5Jcnal4PsG3tbWGKW1yAvcXhyMXf0fN6ZGpVRZFnB+FJqDyBxvsCPEXKhw=="], - "@napi-rs/tar-linux-arm-gnueabihf": ["@napi-rs/tar-linux-arm-gnueabihf@1.1.0", "", { "os": "linux", "cpu": "arm" }, "sha512-SLgIQo3f3EjkZ82ZwvrEgFvMdDAhsxCYjyoSuWfHCz0U16qx3SuGCp8+FYOPYCECHN3ZlGjXnoAIt9ERd0dEUg=="], + "@napi-rs/tar-linux-arm-gnueabihf": ["@napi-rs/tar-linux-arm-gnueabihf@1.1.1", "", { "os": "linux", "cpu": "arm" }, "sha512-an1bJdfyhI5FpZYyTQ20mrqwR+a676i8GkaYc4Uy12dH/a7TJIfrK6Qa2Gm46arZvxUvx56qxoRKXbpOjUPvwA=="], - "@napi-rs/tar-linux-arm64-gnu": ["@napi-rs/tar-linux-arm64-gnu@1.1.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-d014cdle52EGaH6GpYTQOP9Py7glMO1zz/+ynJPjjzYFSxvdYx0byrjumZk2UQdIyGZiJO2MEFpCkEEKFSgPYA=="], + "@napi-rs/tar-linux-arm64-gnu": ["@napi-rs/tar-linux-arm64-gnu@1.1.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-w++Vtx36T2yHTKws7GVnmHHcUT1ybB59xLWSh9A8bwEpJVG4dG7Qub9mFe5cpcbfrJ+XP2mKKxC3oUJSunK3iQ=="], - "@napi-rs/tar-linux-arm64-musl": ["@napi-rs/tar-linux-arm64-musl@1.1.0", "", { "os": "linux", "cpu": "arm64" }, "sha512-L/y1/26q9L/uBqiW/JdOb/Dc94egFvNALUZV2WCGKQXc6UByPBMgdiEyW2dtoYxYYYYc+AKD+jr+wQPcvX2vrQ=="], + "@napi-rs/tar-linux-arm64-musl": ["@napi-rs/tar-linux-arm64-musl@1.1.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-Rh6UFhNtj3i4deJHOBINFIeRL0072mgbeyuK5rl1HokKnNoMKx8qKIZNEzBTTqpogMfDHWGvzyTQdnVxes5dpA=="], - "@napi-rs/tar-linux-ppc64-gnu": ["@napi-rs/tar-linux-ppc64-gnu@1.1.0", "", { "os": "linux", "cpu": "ppc64" }, "sha512-EPE1K/80RQvPbLRJDJs1QmCIcH+7WRi0F73+oTe1582y9RtfGRuzAkzeBuAGRXAQEjRQw/RjtNqr6UTJ+8UuWQ=="], + "@napi-rs/tar-linux-ppc64-gnu": ["@napi-rs/tar-linux-ppc64-gnu@1.1.1", "", { "os": "linux", "cpu": "ppc64" }, "sha512-Cp+AxFbv9zcyAXtnzQi0OzmgDnQgy2w9D4Ubr+iwzMtVgJcztzcEoCcCrN1k2ATdEB01LX2Vb49IaocGOZhC9Q=="], - "@napi-rs/tar-linux-s390x-gnu": ["@napi-rs/tar-linux-s390x-gnu@1.1.0", "", { "os": "linux", "cpu": "s390x" }, "sha512-B2jhWiB1ffw1nQBqLUP1h4+J1ovAxBOoe5N2IqDMOc63fsPZKNqF1PvO/dIem8z7LL4U4bsfmhy3gBfu547oNQ=="], + "@napi-rs/tar-linux-s390x-gnu": ["@napi-rs/tar-linux-s390x-gnu@1.1.1", "", { "os": "linux", "cpu": "s390x" }, "sha512-ZyscC3SYKTBWyDRYjLOKAd5TyJ7q0KACRdQ8bWrb3rgrra1CCIJD66CsGTH6Dh0AVSdfLwZ8MfIIXU6+14BMjQ=="], - "@napi-rs/tar-linux-x64-gnu": ["@napi-rs/tar-linux-x64-gnu@1.1.0", "", { "os": "linux", "cpu": "x64" }, "sha512-tbZDHnb9617lTnsDMGo/eAMZxnsQFnaRe+MszRqHguKfMwkisc9CCJnks/r1o84u5fECI+J/HOrKXgczq/3Oww=="], + "@napi-rs/tar-linux-x64-gnu": ["@napi-rs/tar-linux-x64-gnu@1.1.1", "", { "os": "linux", "cpu": "x64" }, "sha512-LlIv+zg4fiOQge9LQX/ieBdRWE2fhVDjCTHxnunZkbugNmdhdelxWf1RpZb/6ZujWpNF4LPu4N/MW7ygg2oYAQ=="], - "@napi-rs/tar-linux-x64-musl": ["@napi-rs/tar-linux-x64-musl@1.1.0", "", { "os": "linux", "cpu": "x64" }, "sha512-dV6cODlzbO8u6Anmv2N/ilQHq/AWz0xyltuXoLU3yUyXbZcnWYZuB2rL8OBGPmqNcD+x9NdScBNXh7vWN0naSQ=="], + "@napi-rs/tar-linux-x64-musl": ["@napi-rs/tar-linux-x64-musl@1.1.1", "", { "os": "linux", "cpu": "x64" }, "sha512-gZBeoKLjanOVj55qk4EMu13P2i9M0SuINmlGQkOxm1niIJofexzddHUYtqO5o/5QqtyL8lADmAcZplLILMLhHA=="], - "@napi-rs/tar-wasm32-wasi": ["@napi-rs/tar-wasm32-wasi@1.1.0", "", { "dependencies": { "@napi-rs/wasm-runtime": "^1.0.3" }, "cpu": "none" }, "sha512-jIa9nb2HzOrfH0F8QQ9g3WE4aMH5vSI5/1NYVNm9ysCmNjCCtMXCAhlI3WKCdm/DwHf0zLqdrrtDFXODcNaqMw=="], + "@napi-rs/tar-wasm32-wasi": ["@napi-rs/tar-wasm32-wasi@1.1.1", "", { "dependencies": { "@emnapi/core": "1.11.2", "@emnapi/runtime": "1.11.2", "@napi-rs/wasm-runtime": "^1.1.6" }, "cpu": "none" }, "sha512-rwtQ1Mdt/ft6g6I54fJzbUeLspl4yTwj6I3UJ6mitKnrN42soJkcDrdh3Y/FGvlpqZTad2YMQ96fGJl3EtAm2Q=="], - "@napi-rs/tar-win32-arm64-msvc": ["@napi-rs/tar-win32-arm64-msvc@1.1.0", "", { "os": "win32", "cpu": "arm64" }, "sha512-vfpG71OB0ijtjemp3WTdmBKJm9R70KM8vsSExMsIQtV0lVzP07oM1CW6JbNRPXNLhRoue9ofYLiUDk8bE0Hckg=="], + "@napi-rs/tar-win32-arm64-msvc": ["@napi-rs/tar-win32-arm64-msvc@1.1.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-30PVp1AehRpfwxmv5wI4cg0yj3WmWBsZ+1QnLGnvEELu7Eu/+dhNU0nrmhI7VfPgLwSRK2eg9DQTB3tP7Wv9bA=="], - "@napi-rs/tar-win32-ia32-msvc": ["@napi-rs/tar-win32-ia32-msvc@1.1.0", "", { "os": "win32", "cpu": "ia32" }, "sha512-hGPyPW60YSpOSgzfy68DLBHgi6HxkAM+L59ZZZPMQ0TOXjQg+p2EW87+TjZfJOkSpbYiEkULwa/f4a2hcVjsqQ=="], + "@napi-rs/tar-win32-ia32-msvc": ["@napi-rs/tar-win32-ia32-msvc@1.1.1", "", { "os": "win32", "cpu": "ia32" }, "sha512-aI3/rmz+izUChiSeaPxcasAOxhf3FpJNuIHMXlxS/vpW+HIxUsSDR5+XV61PEG5DL4L/75iENVUxmSGM5l2yaw=="], - "@napi-rs/tar-win32-x64-msvc": ["@napi-rs/tar-win32-x64-msvc@1.1.0", "", { "os": "win32", "cpu": "x64" }, "sha512-L6Ed1DxXK9YSCMyvpR8MiNAyKNkQLjsHsHK9E0qnHa8NzLFqzDKhvs5LfnWxM2kJ+F7m/e5n9zPm24kHb3LsVw=="], + "@napi-rs/tar-win32-x64-msvc": ["@napi-rs/tar-win32-x64-msvc@1.1.1", "", { "os": "win32", "cpu": "x64" }, "sha512-yJsB2IsrODQVLKbm2Fg1nHiVRbEj49mSPbj4x7JPZWJI0jGVPjohE2Sif0FBbx8OxsVoUODvS0BwksZZ8jl/OA=="], "@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.1.6", "", { "dependencies": { "@tybys/wasm-util": "^0.10.3" }, "peerDependencies": { "@emnapi/core": "^1.7.1", "@emnapi/runtime": "^1.7.1" } }, "sha512-ZLv/JdUfkvOy9eCnnBaGfiO+XimbjebAeO+MRQqD/B+FR1tnRN0tpKSJHRbE8sFfS6aqsXZ67TQjfwfsxULVbg=="], @@ -733,7 +747,7 @@ "@napi-rs/wasm-tools-win32-x64-msvc": ["@napi-rs/wasm-tools-win32-x64-msvc@1.0.1", "", { "os": "win32", "cpu": "x64" }, "sha512-rEAf05nol3e3eei2sRButmgXP+6ATgm0/38MKhz9Isne82T4rPIMYsCIFj0kOisaGeVwoi2fnm7O9oWp5YVnYQ=="], - "@nodable/entities": ["@nodable/entities@2.2.0", "", {}, "sha512-9uGyhaQavEUMC8AIddIjau4NsnsXhou+j5sBAGojCM1oxmQpVKTWR/9JxABD6UAv12vpIms55fPZKFQEhG6uBg=="], + "@nodable/entities": ["@nodable/entities@3.0.0", "", {}, "sha512-8L9xFeTYKhm49xfIypoe2W5wV1m/3Z58kT+7kR9A8OyFxcPduI4VmxaUMQyKYrRjUoLLSXv6EKKID5Tvj9cUVw=="], "@octokit/auth-token": ["@octokit/auth-token@6.0.0", "", {}, "sha512-P4YJBPdPSpWTQ1NU4XYdvHvXJJDxM6YwpS0FZHRgP7YFkdVxsWcpWGy/NVqlAA7PcPCnMacXlRm1y2PFZRWL/w=="], @@ -799,6 +813,12 @@ "@opentelemetry/core": ["@opentelemetry/core@2.9.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw=="], + "@opentelemetry/exporter-logs-otlp-proto": ["@opentelemetry/exporter-logs-otlp-proto@0.220.0", "", { "dependencies": { "@opentelemetry/otlp-exporter-base": "0.220.0", "@opentelemetry/otlp-transformer": "0.220.0", "@opentelemetry/sdk-logs": "0.220.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-8LZAxdJ0ENDAFwr4j0oY35mHBltiSzvlhdQAPGiC7p9VnxtuSq4SW1gfBAdW6t6hiQG6OwUl8w7KHaOdJPKHWg=="], + + "@opentelemetry/exporter-metrics-otlp-http": ["@opentelemetry/exporter-metrics-otlp-http@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-exporter-base": "0.220.0", "@opentelemetry/otlp-transformer": "0.220.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-metrics": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-Yqt3RBw/bRVncaE9qIIhk4WfjbAQqXuP9FgAaU+IKPndnLEp/cUqZlSC324+bpmduRz7DoTjig8Ub0PeILWXUA=="], + + "@opentelemetry/exporter-metrics-otlp-proto": ["@opentelemetry/exporter-metrics-otlp-proto@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/exporter-metrics-otlp-http": "0.220.0", "@opentelemetry/otlp-exporter-base": "0.220.0", "@opentelemetry/otlp-transformer": "0.220.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-metrics": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-lyO+IQBdSvqHN/ZOW/OzrSWemtfD+HgWngn+HBNLhjy0YrCQQTz0OE/kSekH2Pl340dn9DWzhqHdz5Eftr+HLA=="], + "@opentelemetry/exporter-trace-otlp-proto": ["@opentelemetry/exporter-trace-otlp-proto@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-exporter-base": "0.220.0", "@opentelemetry/otlp-transformer": "0.220.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-trace": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-voTAD8XgJxlK7zLkXh8EzMB09zrQr3tyY/BsnDTlDiQU/UdK58MZ63A3mUjdEDrxMjCVmBHU3WQJhRmQe+Dvzg=="], "@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-transformer": "0.220.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CXYo8UD5Mn9YbgebO2EL4wejtA+gxLmLiu6HCk2KH2BR7XhFN6/6p1UlCb23DYCjeYkndevLHuejCCN1yx4+OQ=="], @@ -877,35 +897,35 @@ "@so-ric/colorspace": ["@so-ric/colorspace@1.1.6", "", { "dependencies": { "color": "^5.0.2", "text-hex": "1.0.x" } }, "sha512-/KiKkpHNOBgkFJwu9sh48LkHSMYGyuTcSFK/qMBdnOAlrRJzRSXAOFB5qwzaVQuDl8wAvHVMkaASQDReTahxuw=="], - "@tailwindcss/node": ["@tailwindcss/node@4.3.2", "", { "dependencies": { "@jridgewell/remapping": "^2.3.5", "enhanced-resolve": "5.21.6", "jiti": "^2.7.0", "lightningcss": "1.32.0", "magic-string": "^0.30.21", "source-map-js": "^1.2.1", "tailwindcss": "4.3.2" } }, "sha512-yWP/sqEcBLaD8JuA6zNwxoYKr75qxTioYwlRwekj5Jr/I5GXnoJfjetH/psLUIv74cYTH2lBUEzBkinthoYcBg=="], + "@tailwindcss/node": ["@tailwindcss/node@4.3.3", "", { "dependencies": { "@jridgewell/remapping": "^2.3.5", "enhanced-resolve": "^5.24.1", "jiti": "^2.7.0", "lightningcss": "1.32.0", "magic-string": "^0.30.21", "source-map-js": "^1.2.1", "tailwindcss": "4.3.3" } }, "sha512-/T8IKEsf9VTU6tLjgC7+sv2mOPtQxzE2jMw7u4Tt40Tx+QSZxpzh95/H6cMKoja9XuW7iMdLJYBB0o9G1CaAgg=="], - "@tailwindcss/oxide": ["@tailwindcss/oxide@4.3.2", "", { "optionalDependencies": { "@tailwindcss/oxide-android-arm64": "4.3.2", "@tailwindcss/oxide-darwin-arm64": "4.3.2", "@tailwindcss/oxide-darwin-x64": "4.3.2", "@tailwindcss/oxide-freebsd-x64": "4.3.2", "@tailwindcss/oxide-linux-arm-gnueabihf": "4.3.2", "@tailwindcss/oxide-linux-arm64-gnu": "4.3.2", "@tailwindcss/oxide-linux-arm64-musl": "4.3.2", "@tailwindcss/oxide-linux-x64-gnu": "4.3.2", "@tailwindcss/oxide-linux-x64-musl": "4.3.2", "@tailwindcss/oxide-wasm32-wasi": "4.3.2", "@tailwindcss/oxide-win32-arm64-msvc": "4.3.2", "@tailwindcss/oxide-win32-x64-msvc": "4.3.2" } }, "sha512-z8ZgnzX8gdNoWLBLqBPoh/sjnxkwvf9ZuWjnO0l0yIzbLa5/9S+eC5QxGZKRobVHIC3/1BoMWjHblqWjcgFgag=="], + "@tailwindcss/oxide": ["@tailwindcss/oxide@4.3.3", "", { "optionalDependencies": { "@tailwindcss/oxide-android-arm64": "4.3.3", "@tailwindcss/oxide-darwin-arm64": "4.3.3", "@tailwindcss/oxide-darwin-x64": "4.3.3", "@tailwindcss/oxide-freebsd-x64": "4.3.3", "@tailwindcss/oxide-linux-arm-gnueabihf": "4.3.3", "@tailwindcss/oxide-linux-arm64-gnu": "4.3.3", "@tailwindcss/oxide-linux-arm64-musl": "4.3.3", "@tailwindcss/oxide-linux-x64-gnu": "4.3.3", "@tailwindcss/oxide-linux-x64-musl": "4.3.3", "@tailwindcss/oxide-wasm32-wasi": "4.3.3", "@tailwindcss/oxide-win32-arm64-msvc": "4.3.3", "@tailwindcss/oxide-win32-x64-msvc": "4.3.3" } }, "sha512-krXjAikiaFSPaK/FkAQT5UTx3VormQaiZ5hBFlJZ9UFQGB/rwg1MZIhHAG9smMQRTdyJxP6Qt5MwMtdyU5FWrA=="], - "@tailwindcss/oxide-android-arm64": ["@tailwindcss/oxide-android-arm64@4.3.2", "", { "os": "android", "cpu": "arm64" }, "sha512-WHxqIuHpvZ5VtdX6GTl1Ik/Vp2YuN42Et+0CdeaVd/frQ9jAvGmvR8vLT+jk3e8/Q3x8kECB9+R17pgpp2BulA=="], + "@tailwindcss/oxide-android-arm64": ["@tailwindcss/oxide-android-arm64@4.3.3", "", { "os": "android", "cpu": "arm64" }, "sha512-Y85A2gmPSkl5Ve5qR86GL4HT509cFqQh1aes9p3sSkyTPwt0Pppf3GkwGe4JPACcRYjgJIEhQgM6dBClnr0NYw=="], - "@tailwindcss/oxide-darwin-arm64": ["@tailwindcss/oxide-darwin-arm64@4.3.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-GZypeUY/IDJW3877KeM+O67vbXr3MBnbtEL4aYhNErv/JWZhye2vGSWWG9tB6iiqR2MqRNkY8IOUy4NdSZV26w=="], + "@tailwindcss/oxide-darwin-arm64": ["@tailwindcss/oxide-darwin-arm64@4.3.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-BiaWatpBcERQFDlOjRDpIVXuFK5PJez5SA4JMg6VYZdBYU+qKfV/vqjcIs+IYmtitf1xYQZTwXvU/8y4lfZUGw=="], - "@tailwindcss/oxide-darwin-x64": ["@tailwindcss/oxide-darwin-x64@4.3.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-UIIzmefR6KO1sDU7MzRqAxC8iBpft/VhkGjTjnhoS6k7Z3rQ9wEgA1ODSiyH/tcSYssulNm4Ci3hOeK1jH7ccQ=="], + "@tailwindcss/oxide-darwin-x64": ["@tailwindcss/oxide-darwin-x64@4.3.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-fAeUqfV5ndhxRwai8cXGzdLvul9utWOmeTkv69unv4ZXixjn61Z+p9lCWdwOwA3TYboG3BwdVuN/RDjhBRl0mw=="], - "@tailwindcss/oxide-freebsd-x64": ["@tailwindcss/oxide-freebsd-x64@4.3.2", "", { "os": "freebsd", "cpu": "x64" }, "sha512-GN+uAmcI6DNspnCDwtOAZrTz6oukJnp337qZvxqCGLd3BHBzJpO0ZbTLRvJNdztOeAmTzewewGIMPb0tk2R4WA=="], + "@tailwindcss/oxide-freebsd-x64": ["@tailwindcss/oxide-freebsd-x64@4.3.3", "", { "os": "freebsd", "cpu": "x64" }, "sha512-iyf5bV6+wnAlflVeEy7R25dupxTNECZN5QMI0qNT6eT+EgaGdZcKhGkr5SdoaWiLJ3spLqIY9VCeSGrwmtg4kw=="], - "@tailwindcss/oxide-linux-arm-gnueabihf": ["@tailwindcss/oxide-linux-arm-gnueabihf@4.3.2", "", { "os": "linux", "cpu": "arm" }, "sha512-4ABn7qSbdHRwTiDiuWNegCyb5+2FJ4vKIKc3DmKrvAFw7MU1Lm11dIkTPwUaFdTzc7IsOpDbqBrlh0x6y36U/w=="], + "@tailwindcss/oxide-linux-arm-gnueabihf": ["@tailwindcss/oxide-linux-arm-gnueabihf@4.3.3", "", { "os": "linux", "cpu": "arm" }, "sha512-aAYUprJAJQWWbRrPvtjdroZ56Md+JM8pMiopS6xGEwDfLhqj+2ver2p4nU4Mb3CRqcMmNBjo8KkUgcxhkzVQGQ=="], - "@tailwindcss/oxide-linux-arm64-gnu": ["@tailwindcss/oxide-linux-arm64-gnu@4.3.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-wDgEIGwoM8w8pufh9LVt1PahDgNdKXrLC2qfAnV3vAmococ9RWbxeAw4pxPttd/TsJfwjyLf90Dg1y9y8I6Emw=="], + "@tailwindcss/oxide-linux-arm64-gnu": ["@tailwindcss/oxide-linux-arm64-gnu@4.3.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-nDxldcEENOxZRzC2uu9jrutZdAAQtb+8WWDCSnWL1zvBk1+FN+x6MtDViPB5AJMfttVCUhehGWus3XBPgatM/w=="], - "@tailwindcss/oxide-linux-arm64-musl": ["@tailwindcss/oxide-linux-arm64-musl@4.3.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-J5Nuk0uZQIiMTJj3LEx4sAA9tMFUoXQZFv1J6An+QGYe53HKRJuFDi0rpq/tuouCZeAbOBY3kQ6g8qeD4TUjtA=="], + "@tailwindcss/oxide-linux-arm64-musl": ["@tailwindcss/oxide-linux-arm64-musl@4.3.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-Md44bD6veX/PC5iyF8cDVnw4HBIANZepRZZ7a8DQOvkfo5WUBwcp6iAuCUz23u+4SUkhJlD3eL7hNdW8ezd/kA=="], - "@tailwindcss/oxide-linux-x64-gnu": ["@tailwindcss/oxide-linux-x64-gnu@4.3.2", "", { "os": "linux", "cpu": "x64" }, "sha512-kqCZpSKOBEJO4mz7OqWoofBZeXTAwaVGPj0ErAj7CojmhKpWVWVOnrt9dE8odoIraZq4oj3ausM37kXi+Tow8w=="], + "@tailwindcss/oxide-linux-x64-gnu": ["@tailwindcss/oxide-linux-x64-gnu@4.3.3", "", { "os": "linux", "cpu": "x64" }, "sha512-tx7us1muwOKAKWao2v/GaafFeQboE6aj88vC6ziN2NCGcRm8gWUhwjzg+YdVB1e4boAtdtma4L43onunI6NS4w=="], - "@tailwindcss/oxide-linux-x64-musl": ["@tailwindcss/oxide-linux-x64-musl@4.3.2", "", { "os": "linux", "cpu": "x64" }, "sha512-cixpqbh2toJDmkuCRI68nXA8ZxNmdK9Y+9v5h3MC3ZQKy/0BO8AWzlkWyRM7JAFSGBlfig4YVTPsK6MVgqz1uw=="], + "@tailwindcss/oxide-linux-x64-musl": ["@tailwindcss/oxide-linux-x64-musl@4.3.3", "", { "os": "linux", "cpu": "x64" }, "sha512-SJxX60smvHgasZoBy11dX6YRjXJFovwWBoedhbQPOBzgFWBHGB+TVPWB9BxzR7TTxU8FQZAI2AyiNCMzFm8Img=="], - "@tailwindcss/oxide-wasm32-wasi": ["@tailwindcss/oxide-wasm32-wasi@4.3.2", "", { "dependencies": { "@emnapi/core": "^1.11.1", "@emnapi/runtime": "^1.11.1", "@emnapi/wasi-threads": "^1.2.2", "@napi-rs/wasm-runtime": "^1.1.4", "@tybys/wasm-util": "^0.10.2", "tslib": "^2.8.1" }, "cpu": "none" }, "sha512-4ec2Z/LOmRsAgU23CS4xeJfcJlmRg94A/XrbGRCF1gyU/zdDfRLYDVsS+ynSZCmGNxQ1jQriQOKMQeQxBA3Isw=="], + "@tailwindcss/oxide-wasm32-wasi": ["@tailwindcss/oxide-wasm32-wasi@4.3.3", "", { "dependencies": { "@emnapi/core": "^1.11.1", "@emnapi/runtime": "^1.11.1", "@emnapi/wasi-threads": "^1.2.2", "@napi-rs/wasm-runtime": "^1.1.4", "@tybys/wasm-util": "^0.10.2", "tslib": "^2.8.1" }, "cpu": "none" }, "sha512-jx1+rPhY/5Ympkktd656HBWEBLxP7dH06losBLjjf5vgCODXvi9KhtftWcMIwTFIDqBr7cRnQkdLnAG+IOlGvQ=="], - "@tailwindcss/oxide-win32-arm64-msvc": ["@tailwindcss/oxide-win32-arm64-msvc@4.3.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-Zyr/M0+XcYZu3bZrUytc7TXvrk0ftWfl8gN2MwekNDzhqhKRUucMPSeOzM0o0wH5AWOU49BsKRrfKxI2atCPMQ=="], + "@tailwindcss/oxide-win32-arm64-msvc": ["@tailwindcss/oxide-win32-arm64-msvc@4.3.3", "", { "os": "win32", "cpu": "arm64" }, "sha512-3rc292Ca2ceK6Ulcc/bAVnTs/3nDtoPhyEKlgPv+yQJQi/JS/AMJlqzxvlDacL1nekbrcf6bTqp/jV4qgnPxNQ=="], - "@tailwindcss/oxide-win32-x64-msvc": ["@tailwindcss/oxide-win32-x64-msvc@4.3.2", "", { "os": "win32", "cpu": "x64" }, "sha512-QI9BO7KlNZsp2GuO0jwAAj5jCDABOKXRkCk2XuKTSaNEFSdfzqswYVTtCHBNKHLsqyjFyFkqlDiwkNbTYSssMQ=="], + "@tailwindcss/oxide-win32-x64-msvc": ["@tailwindcss/oxide-win32-x64-msvc@4.3.3", "", { "os": "win32", "cpu": "x64" }, "sha512-yJ0pwIVc/nYeGoV02WtsN8KYyLQv7kyI2wDnkezyJlGGjkd4QLwDGAwl47YpPJeuI0M0ObaXGSPjvWDPeTPggw=="], - "@tailwindcss/vite": ["@tailwindcss/vite@4.3.2", "", { "dependencies": { "@tailwindcss/node": "4.3.2", "@tailwindcss/oxide": "4.3.2", "tailwindcss": "4.3.2" }, "peerDependencies": { "vite": "^5.2.0 || ^6 || ^7 || ^8" } }, "sha512-eHpMeX4JXfVNJDEcsouTeCBubJBTcTLigeaw/NTUW6PB5ATKKXdyonnXgTBX2VuRbjz1hjfz6C5XAhr52ImQXA=="], + "@tailwindcss/vite": ["@tailwindcss/vite@4.3.3", "", { "dependencies": { "@tailwindcss/node": "4.3.3", "@tailwindcss/oxide": "4.3.3", "tailwindcss": "4.3.3" }, "peerDependencies": { "vite": "^5.2.0 || ^6 || ^7 || ^8" } }, "sha512-yYU8cogLeSh/ms2jh8Fj7jaba/EWa7Ja6GoUqYZaraEuCI5YS6ms6ObZgjjedm+jm6XZjdNRWBpPP6Z86oOxcw=="], "@ts-morph/common": ["@ts-morph/common@0.29.0", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, "sha512-35oUmphHbJvQ/+UTwFNme/t2p3FoKiGJ5auTjjpNTop2dyREspirjMy82PLSC1pnDJ8ah1GU98hwpVt64YXQsg=="], @@ -1003,8 +1023,6 @@ "adm-zip": ["adm-zip@0.5.18", "", {}, "sha512-ufJnssQGbxzLNS1Ho9bCtX4rQKCCvoVuDLHoJyc3F9dOGDB4BkWs2Ci0kv53lqocAEQ/Cbi+I2XCsNYGqVYqng=="], - "ansi-escapes": ["ansi-escapes@7.3.0", "", { "dependencies": { "environment": "^1.0.0" } }, "sha512-BvU8nYgGQBxcmMuEeUEmNTvrMVjJNSH7RgW24vXexN4Ven6qCvy4TntnvlnwnMLTVlcRQQdbRY8NKnaIoeWDNg=="], - "ansi-regex": ["ansi-regex@6.2.2", "", {}, "sha512-Bq3SmSpyFHaWjPk8If9yc6svM8c56dB5BAtW4Qbw5jHTwwXXcTLoRMkpDJp6VL0XzlWaCHTXrkFURMYmD0sLqg=="], "ansi-styles": ["ansi-styles@6.2.3", "", {}, "sha512-4Dj6M28JB+oAH8kFkTLUo+a2jwOFkuqb3yucU0CANcRRUbxS0cP0nZYCGjcc3BNXwRIsUVmDGgzawme7zvJHvg=="], @@ -1045,7 +1063,7 @@ "callsites": ["callsites@3.1.0", "", {}, "sha512-P8BjAsXvZS+VIDUI11hHCQEv74YT67YUi5JJFNWIqL235sBmjX4+qx9Muvls5ivyNENctx46xQLQ3aTuE7ssaQ=="], - "caniuse-lite": ["caniuse-lite@1.0.30001805", "", {}, "sha512-52noaS3DubycKSXaU30TwPGIp+POyQSUVa5jBEq3vkRkY0kjyb3LQgvhU6WGyCcyXqVLWO0Cw0Q6BSdD0kUfVA=="], + "caniuse-lite": ["caniuse-lite@1.0.30001806", "", {}, "sha512-72Cuvd95zbSYPKq6Fhg8eDJRlzgWDf7/mtoZv6Qe/DYNCEBdNxoA3+rZAU2ZhGCpZlns3EssFavaZomckT5Uuw=="], "chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="], @@ -1057,12 +1075,8 @@ "chromium-bidi": ["chromium-bidi@16.0.1", "", { "dependencies": { "mitt": "^3.0.1", "zod": "^3.24.1" }, "peerDependencies": { "devtools-protocol": "*" } }, "sha512-J63PGu/9PpeCwLIcKYyzWP6yaVL5pxuBc0shlYCYM8BaAkmlwiQboXO1iNbOgSDbVklEyYFfNEcHD8oOAWacUA=="], - "cli-cursor": ["cli-cursor@5.0.0", "", { "dependencies": { "restore-cursor": "^5.0.0" } }, "sha512-aCj4O5wKyszjMmDT4tZj93kxyydN/K5zPWSCe6/0AV/AA1pqe5ZBIw0a2ZfPQV7lL5/yb5HsUreJ6UFAF1tEQw=="], - "cli-progress": ["cli-progress@3.12.0", "", { "dependencies": { "string-width": "^4.2.3" } }, "sha512-tRkV3HJ1ASwm19THiiLIXLO7Im7wlTuKnvkYaTkyoAPefqjNg7W7DHKUlGRxy9vxDvbyCYQkQozvptuMkGCg8A=="], - "cli-truncate": ["cli-truncate@5.2.0", "", { "dependencies": { "slice-ansi": "^8.0.0", "string-width": "^8.2.0" } }, "sha512-xRwvIOMGrfOAnM1JYtqQImuaNtDEv9v6oIYAs4LIHwTiKee8uwvIi363igssOC0O5U04i4AlENs79LQLu9tEMw=="], - "cli-width": ["cli-width@4.1.0", "", {}, "sha512-ouuZd4/dm2Sw5Gmqy6bGyNNNe1qt9RpmxveLSO7KcgsTnU7RXfsw+/bukWGo1abgBiMAic068rclZsO4IWmmxQ=="], "clipanion": ["clipanion@4.0.0-rc.4", "", { "dependencies": { "typanion": "^3.8.0" } }, "sha512-CXkMQxU6s9GklO/1f714dkKBMu1lopS1WFF0B8o4AxPykR1hpozxSiUZ5ZUeBjfPgCWqbcNOtZVFhB8Lkfp1+Q=="], @@ -1145,7 +1159,7 @@ "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], - "electron-to-chromium": ["electron-to-chromium@1.5.389", "", {}, "sha512-cEto7aeOqBfU1D+c5py5pE+ooscKE75JifxLBdFUZsqAxRS6y7kebtxAZvICszSl05gPjYHDTjY+lXpyGvpJbg=="], + "electron-to-chromium": ["electron-to-chromium@1.5.393", "", {}, "sha512-kiDJdIUawuEIcp9XoICKp1iTYDEbgguIPq526N1Q7jIQDeQ3CqoMx71025PI/7E48Ddtw2HuWsVjY7afEgNxmg=="], "emnapi": ["emnapi@1.11.2", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-iMt/XQc69fFn2EvcU6tm14HmXKwyy0lnABugsQlqp6xFuZIUuO+ONVSg2mz+MTVF8WbC+bic65AvRXdoldALKg=="], @@ -1153,12 +1167,10 @@ "enabled": ["enabled@2.0.0", "", {}, "sha512-AKrN98kuwOzMIdAizXGI86UFBoo26CL21UM763y1h/GMSJ4/OHU9k2YlsmBpyScFo/wbLzWQJBMCW4+IO3/+OQ=="], - "enhanced-resolve": ["enhanced-resolve@5.21.6", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.3" } }, "sha512-aNnGCvbJ/RIyWo1IuhNdVjnNF+EjH9wpzpNHt+ci/m9He9LJvUN8wrCcXjp9cWsGNAuvSpVFTx/vraAFQ8qGjQ=="], + "enhanced-resolve": ["enhanced-resolve@5.24.2", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.3" } }, "sha512-rpsZEGT1jFuve6QlpyRp9ckQ+kN61hvF9BzCPyMdaKTm8UJce96KBn3sorXOFXlzjPrs3Vc4T1NsSroZ3PxlFw=="], "entities": ["entities@7.0.1", "", {}, "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA=="], - "environment": ["environment@1.1.0", "", {}, "sha512-xUtoPkMggbz0MPyPiIWr1Kp4aeWJjDZ6SMvURhimjdZgsRuDplF5/s9hcgGhyXMhs+6vpnuoiZ2kFiu3FMnS8Q=="], - "es-define-property": ["es-define-property@1.0.1", "", {}, "sha512-e3nRfgfUZ4rNGL232gUgX06QNyyez04KdjFrF+LTRoOXmrOgFKDg4BCdsjW8EnT69eqdYGmRpJwiPVYNrCaW3g=="], "es-errors": ["es-errors@1.3.0", "", {}, "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw=="], @@ -1171,8 +1183,6 @@ "escape-string-regexp": ["escape-string-regexp@4.0.0", "", {}, "sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA=="], - "eventemitter3": ["eventemitter3@5.0.4", "", {}, "sha512-mlsTRyGaPBjPedk6Bvw+aqbsXDtoAyAzm5MO7JgU+yVRyMQ5O8bD4Kcci7BS85f93veegeCPkL8R4GLClnjLFw=="], - "fast-string-truncated-width": ["fast-string-truncated-width@3.0.3", "", {}, "sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g=="], "fast-string-width": ["fast-string-width@3.0.2", "", { "dependencies": { "fast-string-truncated-width": "^3.0.2" } }, "sha512-gX8LrtNEI5hq8DVUfRQMbr5lpaS4nMIWV+7XEbXk2b8kiQIizgnlr12B4dA3ZEx3308ze0O4Q1R+cHts8kyUJg=="], @@ -1181,7 +1191,7 @@ "fast-xml-builder": ["fast-xml-builder@1.3.0", "", { "dependencies": { "path-expression-matcher": "^1.6.2", "xml-naming": "^0.3.0" } }, "sha512-F74cZEdCvuw9P41GAC3rod4X04jjWGM1JPEv/GWSqFTWLsdyMSBMBMlm9Hk3GLBgLBbdBNY8yee0pQh2RBVESQ=="], - "fast-xml-parser": ["fast-xml-parser@5.10.0", "", { "dependencies": { "@nodable/entities": "^2.2.0", "fast-xml-builder": "^1.2.0", "is-unsafe": "^2.0.0", "path-expression-matcher": "^1.6.2", "strnum": "^2.4.1", "xml-naming": "^0.3.0" }, "bin": { "fxparser": "src/cli/cli.js" } }, "sha512-SLhnTEqE5QpJHq/6zl9bsmImEP2adv+y6Wy+cJa7nVTRzQh1OZfCe9k29M5xN74LWnu0xa1zrUrq3KnOKl92Fg=="], + "fast-xml-parser": ["fast-xml-parser@5.10.1", "", { "dependencies": { "@nodable/entities": "^3.0.0", "fast-xml-builder": "^1.2.0", "is-unsafe": "^2.0.0", "path-expression-matcher": "^1.6.2", "strnum": "^2.4.1", "xml-naming": "^0.3.0" }, "bin": { "fxparser": "src/cli/cli.js" } }, "sha512-IEMIf7298kXuZSRFoGfMYrl7is8LpavODgbNz1cwIudv7KwVFnuU+UsMporfq6PD6aXSlawZlARiA3UywCTfMw=="], "fastembed": ["fastembed@2.1.0", "", { "dependencies": { "@anush008/tokenizers": "^0.0.0", "@huggingface/hub": "^2.7.1", "onnxruntime-node": "1.21.0", "progress": "^2.0.3", "tar": "^6.2.0" } }, "sha512-oQkpcRHBppJ3+a3w9dU0uytSY0N1cnEa/iVMc8AXEd+tvT529GekOEFhNviJy89R3lvQXF6cdIMTXHj1Gi00xQ=="], @@ -1243,7 +1253,7 @@ "internmap": ["internmap@2.0.3", "", {}, "sha512-5Hh7Y1wQbvY5ooGgPbDaL5iYLAPzMTUrjMulskHLH6wnv/A+1q5rgEaiuqEjB+oxGXIVZs1FF+R/KPN3ZSQYYg=="], - "is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], + "is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="], "is-obj": ["is-obj@2.0.0", "", {}, "sha512-drqDG3cbczxxEJRoOXcOjtdp1J/lyp1mNn0xaznRs8+muBhgQcrnbspox5X5fOw0HnMnbfDzvnEMEtqDEJEo8w=="], @@ -1301,14 +1311,10 @@ "linkedom": ["linkedom@0.18.13", "", { "dependencies": { "css-select": "^7.0.0", "cssom": "^0.5.0", "html-escaper": "^3.0.3", "htmlparser2": "^10.1.0", "uhyphen": "^0.2.0" }, "peerDependencies": { "canvas": ">= 2" }, "optionalPeers": ["canvas"] }, "sha512-ES/o9qotMpzpN2MHs+Iq/JcVoOj8Fa5wiQYrTdFpvAnwXL0g66XHHUc9WUMk6nAlBtGsFQ24ne+SYnvnaQ2FSw=="], - "lint-staged": ["lint-staged@17.0.8", "", { "dependencies": { "listr2": "^10.2.1", "picomatch": "^4.0.4", "string-argv": "^0.3.2", "tinyexec": "^1.2.4" }, "optionalDependencies": { "yaml": "^2.9.0" }, "bin": { "lint-staged": "bin/lint-staged.js" } }, "sha512-B2P/d+jVW0UXOQ0MVMLrB/9ydA1P+zz6jYfdrbbEd9ur3S2rcbduFWKiUCC02Sm5hbC8nrm7y24WuYMG54HfxA=="], - - "listr2": ["listr2@10.2.2", "", { "dependencies": { "cli-truncate": "^5.2.0", "eventemitter3": "^5.0.4", "log-update": "^6.1.0", "rfdc": "^1.4.1", "wrap-ansi": "^10.0.0" } }, "sha512-JtNtbZj8q5BnDMR7trpwvwk3RIrANtIVzEUm8w7amp6xelLgyuq+4WZoTH913XaQAoH/cNdYhaNzBPA2U3xbDw=="], + "lint-staged": ["lint-staged@17.1.0", "", { "dependencies": { "picomatch": "^4.0.5", "string-argv": "^0.3.2", "tinyexec": "^1.2.4" }, "optionalDependencies": { "yaml": "^2.9.0" }, "bin": { "lint-staged": "bin/lint-staged.js" } }, "sha512-d7UQRu/9ZPgfu4+hu/k0wny5GEaIxo+2jb2LJqQDkE7cHRTm1HGqNUDq5UOwsGPpjpaNAFmgAsYo3TR+i9cSJw=="], "lodash.isequal": ["lodash.isequal@4.5.0", "", {}, "sha512-pDo3lu8Jhfjqls6GkMgpahsF9kCyayhgykjyLMNFTKWrpVdAQtYyB4muAMWozBB4ig/dtWAmsMxLEI8wuz+DYQ=="], - "log-update": ["log-update@6.1.0", "", { "dependencies": { "ansi-escapes": "^7.0.0", "cli-cursor": "^5.0.0", "slice-ansi": "^7.1.0", "strip-ansi": "^7.1.0", "wrap-ansi": "^9.0.0" } }, "sha512-9ie8ItPR6tjY5uYJh8K/Zrv/RMZ5VOlOWvtZdEHYSTFKZfIBPQa9tOAEeAWhd+AnIneLJ22w5fjOYtoutpWq5w=="], - "logform": ["logform@2.7.0", "", { "dependencies": { "@colors/colors": "1.6.0", "@types/triple-beam": "^1.3.2", "fecha": "^4.2.0", "ms": "^2.1.1", "safe-stable-stringify": "^2.3.1", "triple-beam": "^1.3.0" } }, "sha512-TFYA4jnP7PVbmlBIfhlSe+WKxs9dklXMTEGcBCIvLhE/Tn3H6Gk1norupVW7m5Cnd4bLcr08AytbyV/xj7f/kQ=="], "long": ["long@5.3.2", "", {}, "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA=="], @@ -1317,7 +1323,7 @@ "lru-cache": ["lru-cache@11.5.2", "", {}, "sha512-4pfM1Ff0x50o0tQwb5ucw/RzNyD0/YJME6IVcStalZuMWxdt3sR3huStTtxz4PUmvZfRguvDejasvQ2kifR11g=="], - "lucide-react": ["lucide-react@1.24.0", "", { "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, "sha512-YT6mBD8lGKkg4nM39enlm94/sfJIiW0YKUT60fBy4YK8tai31ylg1VhGNWxkpSKHo9UagfnZqwIff3HTDQwXeA=="], + "lucide-react": ["lucide-react@1.25.0", "", { "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, "sha512-/mdJTRbiwcLOQ1NZZK1amZF9rIZyvO18D6r9TngE6TG1NmqHgFuT4eE7Xrkm9UsXMbBJD1NlfwHVltCDWHrOTw=="], "magic-string": ["magic-string@0.30.21", "", { "dependencies": { "@jridgewell/sourcemap-codec": "^1.5.5" } }, "sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ=="], @@ -1329,8 +1335,6 @@ "merge-anything": ["merge-anything@5.1.7", "", { "dependencies": { "is-what": "^4.1.8" } }, "sha512-eRtbOb1N5iyH0tkQDAoQ4Ipsp/5qSR79Dzrz8hEPxRX10RWWR/iQXdoKmBSRCThY1Fh5EhISDtpSc93fpxUniQ=="], - "mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="], - "minimatch": ["minimatch@10.2.5", "", { "dependencies": { "brace-expansion": "^5.0.5" } }, "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg=="], "minimist": ["minimist@1.2.8", "", {}, "sha512-2yyAR8qBkN3YuheJanUpWC5U3bb5osDywNB8RzDVlDwDHbocAJveqqj1u8+SVD7jkWT4yvsHCpWqqWqAxb0zCA=="], @@ -1343,7 +1347,7 @@ "mkdirp": ["mkdirp@1.0.4", "", { "bin": { "mkdirp": "bin/cmd.js" } }, "sha512-vVqVZQyf3WLx2Shd0qJ9xuvqgAyKPLAiqITEtqW0oIUjzo3PePDd6fW9iFz30ef7Ysp/oiWqbhszeGWW2T6Gzw=="], - "modern-tar": ["modern-tar@0.7.6", "", {}, "sha512-sweCIVXzx1aIGTCdzcMlSZt1h8k5Tmk08VNAuRk3IU28XamGiOH5ypi11g6De2CH7PhYqSSnGy2A/EFhbWnVKg=="], + "modern-tar": ["modern-tar@0.7.7", "", {}, "sha512-t9VmxaqrmANnEOBhpSDI6HD192Ge48k8vmWqQQL7hSFEqHEYwZbbsu49+aKLWZeRvFs3j1pMhXOqqF4kPlvjkQ=="], "moment": ["moment@2.30.1", "", {}, "sha512-uEmtNhbDOrWPFS+hdjFCBfy9f2YoyzRpwcl+DqpC6taX21FzsTLQVbMV/W7PzNSX6x/bhC1zA3c2UQ5NzH6how=="], @@ -1371,12 +1375,10 @@ "object-keys": ["object-keys@1.1.1", "", {}, "sha512-NuAESUOUMrlIXOfHKzD6bpPu3tYt3xvjNdRIQ+FeT0lNb4K8WR70CaDxhuNguS2XG+GjkyMwOzsN5ZktImfhLA=="], - "obug": ["obug@2.1.3", "", {}, "sha512-9miFgM2OFba7hB+pRgvtV84pYTBaoTHohvmIgiRt6dRIzbwEOIaNaP+dIlGs2fNFoB0SeISs0Jz5WFVRid6Xyg=="], + "obug": ["obug@2.1.4", "", {}, "sha512-4a+OsYv9UktOJKE+l1A4OufDgdRF9PifWj+tJnHURo/P+WOxpG4GzUFL9qCalmWauao6ogiG+QvnCovwPoyAWA=="], "one-time": ["one-time@1.0.0", "", { "dependencies": { "fn.name": "1.x.x" } }, "sha512-5DXOiRKwuSEcQ/l0kGCF6Q3jcADFv5tSmRaJck/OqkVFcOzutB134KRSfF0xDrL39MNnqxbHBbUUcjZIhTgb2g=="], - "onetime": ["onetime@7.0.0", "", { "dependencies": { "mimic-function": "^5.0.0" } }, "sha512-VXJjc87FScF88uafS3JllDgvAm+c/Slfz06lorj2uAY34rlUu0Nt+v8wreiImcrgAjjIHp1rXpTDlLOGw29WwQ=="], - "onnxruntime-common": ["onnxruntime-common@1.26.0", "", {}, "sha512-qVyMR4lcWgbkc4getFV+GQijsTnbg/siteoqcDwa3sI/LxbrMSNw4ePyvCq/ymdQaRomCA7YuWmhzsswxvymdw=="], "onnxruntime-node": ["onnxruntime-node@1.26.0", "", { "dependencies": { "adm-zip": "^0.5.16", "global-agent": "^4.1.3", "onnxruntime-common": "1.26.0" }, "os": [ "linux", "win32", "darwin", ] }, "sha512-OHl6PiOEOqxaLHL0N9eFrbzS7IGmu3BtJNH3RTEnRAheCIkfc3gjcjl4sGcjp9C22ZC9YTquDOxSdT/stBQ6BQ=="], @@ -1403,7 +1405,7 @@ "platform": ["platform@1.3.6", "", {}, "sha512-fnWVljUchTro6RiCFvCXBbNhJc2NijN7oIQxbwsyL0buWJPG85v81ehlHI9fXrJsMNgTofEoWIQeClKpgxFLrg=="], - "postcss": ["postcss@8.5.17", "", { "dependencies": { "nanoid": "^3.3.12", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" } }, "sha512-J7EF+8X+CzRPaJPOv9Ck2wNWJvGnnl3PcNPAdGg6GTLjyVpyQ0yATMSXRFRV01BviT/9Gwuc3rjEyJbDJG9a4w=="], + "postcss": ["postcss@8.5.19", "", { "dependencies": { "nanoid": "^3.3.12", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" } }, "sha512-Mz8SaolMd8nB+G13WkORcxQKHZ/NE4xXevtkJHVuG+guo9/wYKlIMTKAqGdEmYOXR2ijPjTYNHssizdaVSUNdQ=="], "prettier": ["prettier@3.9.5", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-/FVl766LpUfB5vXgCYOYa0MeV/441Ia99AeICQIQFTY/Nw0roZwULcXpku5i1/m5kt/baz+s4Zogspd839HSMg=="], @@ -1427,10 +1429,6 @@ "regexp-tree": ["regexp-tree@0.1.27", "", { "bin": { "regexp-tree": "bin/regexp-tree" } }, "sha512-iETxpjK6YoRWJG5o6hXLwvjYAoW+FEZn9os0PD/b6AP6xQwsa/Y7lCVgIixBbUPMfhu+i2LtdeAqVTgGlQarfA=="], - "restore-cursor": ["restore-cursor@5.1.0", "", { "dependencies": { "onetime": "^7.0.0", "signal-exit": "^4.1.0" } }, "sha512-oMA2dcrw6u0YfxJQXm342bFKX/E4sG9rbTzO9ptUcR/e8A33cHuvStiYOwH7fszkZlZ1z/ta9AAoPk2F4qIOHA=="], - - "rfdc": ["rfdc@1.4.1", "", {}, "sha512-q1b3N5QkRUWUl7iyylaaj3kOpIT0N2i9MqIEQXP73GVsN9cw3fdx8X63cEmWhJGi2PPCF23Ijp7ktmd39rawIA=="], - "roarr": ["roarr@2.15.4", "", { "dependencies": { "boolean": "^3.0.1", "detect-node": "^2.0.4", "globalthis": "^1.0.1", "json-stringify-safe": "^5.0.1", "semver-compare": "^1.0.0", "sprintf-js": "^1.1.2" } }, "sha512-CHhPh+UNHD2GTXNYhPWLnU8ONHdI+5DI+4EYIAOaiD63rHeYlZvyh8P+in5999TTSFgUYuKUAjzRI4mdh/p+2A=="], "robomp-web": ["robomp-web@workspace:python/robomp/web"], @@ -1477,8 +1475,6 @@ "signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="], - "slice-ansi": ["slice-ansi@8.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "is-fullwidth-code-point": "^5.1.0" } }, "sha512-stxByr12oeeOyY2BlviTNQlYV5xOj47GirPr4yA1hE9JCtxfQN0+tVbkxwCtYDQWhEKWFHsEK48ORg5jrouCAg=="], - "solid-js": ["solid-js@1.9.14", "", { "dependencies": { "csstype": "^3.1.0", "seroval": "~1.5.4", "seroval-plugins": "~1.5.4" } }, "sha512-sAEXC0Kk0S1EDg+8ysEWJDbYhA3RRoEjwuySUGlKIemeo0I5YZfOyumNjNs9Sv3y2nmhD+0rW66ag2HsMuQiGQ=="], "solid-refresh": ["solid-refresh@0.6.3", "", { "dependencies": { "@babel/generator": "^7.23.6", "@babel/helper-module-imports": "^7.22.15", "@babel/types": "^7.23.6" }, "peerDependencies": { "solid-js": "^1.3" } }, "sha512-F3aPsX6hVw9ttm5LYlth8Q15x6MlI/J3Dn+o3EQyRTtTxidepSTwAYdozt01/YA+7ObcciagGEyXIopGZzQtbA=="], @@ -1503,7 +1499,7 @@ "tailwind-merge": ["tailwind-merge@3.6.0", "", {}, "sha512-uxL7qAVQriqRQPAyK3pj66VqskWqoZ37PW94jwOTwNfq/z9oyu1V+eqrZqtR2+fCiXdYOZe/Modt8GtvqNzu+w=="], - "tailwindcss": ["tailwindcss@4.3.2", "", {}, "sha512-WtctNNSH8A9jlMIqxzuYumOHU5uGZyRv0Q5svQl+oEPy5w84YpBxdb7MdqyiSPQge5jTJ6zFQLq0PFygdccSBA=="], + "tailwindcss": ["tailwindcss@4.3.3", "", {}, "sha512-gOhV3P7ufE62QDGg1zVaTgCR+EtPv92k2nIhVcVKcLmxT1sUBsQGhnZj175j+MqRt4zLF7ic+sCYjfhxMxj7YQ=="], "tapable": ["tapable@2.3.3", "", {}, "sha512-uxc/zpqFg6x7C8vOE7lh6Lbda8eEL9zmVm/PLeTPBRhh1xCgdWaQ+J1CUieGpIfm2HdtsUpRv+HshiasBMcc6A=="], @@ -1549,9 +1545,9 @@ "vali-date": ["vali-date@1.0.0", "", {}, "sha512-sgECfZthyaCKW10N0fm27cg8HYTFK5qMWgypqkXMQ4Wbl/zZKx7xZICgcoxIIE+WFAP/MBL2EFwC/YvLxw3Zeg=="], - "vite": ["vite@8.1.4", "", { "dependencies": { "lightningcss": "^1.32.0", "picomatch": "^4.0.5", "postcss": "^8.5.16", "rolldown": "~1.1.4", "tinyglobby": "^0.2.17" }, "optionalDependencies": { "fsevents": "~2.3.3" }, "peerDependencies": { "@types/node": "^20.19.0 || >=22.12.0", "@vitejs/devtools": "^0.3.0", "esbuild": "^0.27.0 || ^0.28.0", "jiti": ">=1.21.0", "less": "^4.0.0", "sass": "^1.70.0", "sass-embedded": "^1.70.0", "stylus": ">=0.54.8", "sugarss": "^5.0.0", "terser": "^5.16.0", "tsx": "^4.8.1", "yaml": "^2.4.2" }, "optionalPeers": ["@types/node", "@vitejs/devtools", "esbuild", "jiti", "less", "sass", "sass-embedded", "stylus", "sugarss", "terser", "tsx", "yaml"], "bin": { "vite": "bin/vite.js" } }, "sha512-bTT9PsdWO+MQMNG9ZXIP/qM9wGh37DFxTV/sPq9cFpHr3w4jkgef032PkAL9jAqhk3Nz8NQw3O8n6/xFkqO4QQ=="], + "vite": ["vite@8.1.5", "", { "dependencies": { "lightningcss": "^1.32.0", "picomatch": "^4.0.5", "postcss": "^8.5.17", "rolldown": "~1.1.5", "tinyglobby": "^0.2.17" }, "optionalDependencies": { "fsevents": "~2.3.3" }, "peerDependencies": { "@types/node": "^20.19.0 || >=22.12.0", "@vitejs/devtools": "^0.3.0", "esbuild": "^0.27.0 || ^0.28.0", "jiti": ">=1.21.0", "less": "^4.0.0", "sass": "^1.70.0", "sass-embedded": "^1.70.0", "stylus": ">=0.54.8", "sugarss": "^5.0.0", "terser": "^5.16.0", "tsx": "^4.8.1", "yaml": "^2.4.2" }, "optionalPeers": ["@types/node", "@vitejs/devtools", "esbuild", "jiti", "less", "sass", "sass-embedded", "stylus", "sugarss", "terser", "tsx", "yaml"], "bin": { "vite": "bin/vite.js" } }, "sha512-7ULLwsCdYx/nRyrpiEwvqb5TFHrMVZyBt+rg/OAXT7rgj/z+DtTDyKFeLAdDkubDVDKD8jOsndmy7m55XcfUsw=="], - "vite-plugin-solid": ["vite-plugin-solid@2.11.12", "", { "dependencies": { "@babel/core": "^7.23.3", "@types/babel__core": "^7.20.4", "babel-preset-solid": "^1.8.4", "merge-anything": "^5.1.7", "solid-refresh": "^0.6.3", "vitefu": "^1.0.4" }, "peerDependencies": { "@testing-library/jest-dom": "^5.16.6 || ^5.17.0 || ^6.*", "solid-js": "^1.7.2", "vite": "^3.0.0 || ^4.0.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0" }, "optionalPeers": ["@testing-library/jest-dom"] }, "sha512-FgjPcx2OwX9h6f28jli7A4bG7PP3te8uyakE5iqsmpq3Jqi1TWLgSroC9N6cMfGRU2zXsl4Q6ISvTr2VL0QHpA=="], + "vite-plugin-solid": ["vite-plugin-solid@2.11.13", "", { "dependencies": { "@babel/core": "^7.23.3", "@types/babel__core": "^7.20.4", "babel-preset-solid": "^1.8.4", "merge-anything": "^5.1.7", "solid-refresh": "^0.6.3", "vitefu": "^1.0.4" }, "peerDependencies": { "@testing-library/jest-dom": "^5.16.6 || ^5.17.0 || ^6.*", "solid-js": "^1.7.2", "vite": "^3.0.0 || ^4.0.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0 || ^9.0.0" }, "optionalPeers": ["@testing-library/jest-dom"] }, "sha512-YaCNMzwIawUO8K16uj5jaxUJIYCstBEjkUppC5OKElBz/K2+R7Q7MWycHDgqzjBca5WWy/2ZQQ45IHexaabYew=="], "vitefu": ["vitefu@1.1.3", "", { "peerDependencies": { "vite": "^3.0.0 || ^4.0.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0" }, "optionalPeers": ["vite"] }, "sha512-ub4okH7Z5KLjb6hDyjqrGXqWtWvoYdU3IGm/NorpgHncKoLTCfRIbvlhBm7r0YstIaQRYlp4yEbFqDcKSzXSSg=="], @@ -1565,9 +1561,9 @@ "wordwrap": ["wordwrap@1.0.0", "", {}, "sha512-gvVzJFlPycKc5dZN4yPkP8w7Dc37BtP1yczEneOb4uq34pXZcvrtRTmWV8W+Ume+XCxKgbjM+nevkyFPMybd4Q=="], - "wrap-ansi": ["wrap-ansi@10.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "string-width": "^8.2.0", "strip-ansi": "^7.1.2" } }, "sha512-SGcvg80f0wUy2/fXES19feHMz8E0JoXv2uNgHOu4Dgi2OrCy1lqwFYEJz1BLbDI0exjPMe/ZdzZ/YpGECBG/aQ=="], + "wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="], - "ws": ["ws@8.21.0", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g=="], + "ws": ["ws@8.21.1", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-+0NTnW77fFN/DjQi6k/Sq/Yvk4Sgajw7urW8V+asjXnRgDs9gyGkdb7EzgfhA4goXsRIZKE28fzIXBHEzhuiWw=="], "xml-naming": ["xml-naming@0.3.0", "", {}, "sha512-ghig2TBE/H11aOVgmahA3MhimvkBr6JIYknH/Dhdk10nXwdbIqBJsbfMxpvFPG8bAw77gN29aQWvKpmVoPlvPQ=="], @@ -1597,6 +1593,10 @@ "@isaacs/fs-minipass/minipass": ["minipass@7.1.3", "", {}, "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A=="], + "@napi-rs/lzma-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" } }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], + + "@napi-rs/tar-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" } }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], + "@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], @@ -1617,10 +1617,6 @@ "cli-progress/string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="], - "cli-truncate/string-width": ["string-width@8.2.2", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-GaPUh5gfdrYzqeVNZvUfT23vYYxXzKYidUcnMtJg/3rxRV63EFZy3k6xfKlmfeJD0176lnUV/Usr3XcwSvFzpg=="], - - "cliui/wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="], - "dom-serializer/domelementtype": ["domelementtype@3.0.0", "", {}, "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg=="], "dom-serializer/entities": ["entities@8.0.0", "", {}, "sha512-zwfzJecQ/Uej6tusMqwAqU/6KL2XaB2VZ2Jg54Je6ahNBGNH6Ek6g3jjNCF0fG9EWQKGZNddNjU5F1ZQn/sBnA=="], @@ -1641,10 +1637,6 @@ "jszip/readable-stream": ["readable-stream@2.3.8", "", { "dependencies": { "core-util-is": "~1.0.0", "inherits": "~2.0.3", "isarray": "~1.0.0", "process-nextick-args": "~2.0.0", "safe-buffer": "~5.1.1", "string_decoder": "~1.1.1", "util-deprecate": "~1.0.1" } }, "sha512-8p0AUk4XODgIewSi0l8Epjs+EVnWiK7NoDIEGU0HhE7+ZyY8D1IMY7odu5lRrFXGg71L15KG8QrPmum45RTtdA=="], - "log-update/slice-ansi": ["slice-ansi@7.1.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "is-fullwidth-code-point": "^5.0.0" } }, "sha512-iOBWFgUX7caIZiuutICxVgX1SdxwAVFFKwt1EvMYYec/NWO5meOJ6K5uQxhrYBdQJne4KxiqZc+KptFOWFSI9w=="], - - "log-update/wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="], - "minizlib/minipass": ["minipass@3.3.6", "", { "dependencies": { "yallist": "^4.0.0" } }, "sha512-DxiNidxSEK+tHG6zOIklvNOwm3hvCrbUrdtzY74U6HKTJxvIDfOUL5W5P2Ghd3DTkhhKPYGqeNUIh5qcM4YBfw=="], "onnxruntime-web/onnxruntime-common": ["onnxruntime-common@1.24.0-dev.20251116-b39e144322", "", {}, "sha512-BOoomdHYmNRL5r4iQ4bMvsl2t0/hzVQ3OM3PHD0gxeXu1PmggqBv3puZicEUVOA3AtHHYmqZtjMj9FOfGrATTw=="], @@ -1655,8 +1647,6 @@ "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], - "wrap-ansi/string-width": ["string-width@8.2.2", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-GaPUh5gfdrYzqeVNZvUfT23vYYxXzKYidUcnMtJg/3rxRV63EFZy3k6xfKlmfeJD0176lnUV/Usr3XcwSvFzpg=="], - "@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], "@huggingface/transformers/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="], @@ -1665,8 +1655,6 @@ "cli-progress/string-width/emoji-regex": ["emoji-regex@8.0.0", "", {}, "sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A=="], - "cli-progress/string-width/is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="], - "cli-progress/string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], "fastembed/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="], diff --git a/crates/pi-iso/src/linux_reflink.rs b/crates/pi-iso/src/linux_reflink.rs index ef15d806d..cdfe49ce7 100644 --- a/crates/pi-iso/src/linux_reflink.rs +++ b/crates/pi-iso/src/linux_reflink.rs @@ -80,7 +80,9 @@ mod imp { use crate::{IsoError, IsoResult}; - const FICLONE: libc::c_ulong = 0x4004_9409; + // `libc::Ioctl` is `c_int` on musl and `c_ulong` on glibc; the constant fits + // both. + const FICLONE: libc::Ioctl = 0x4004_9409; pub fn start(lower: &Path, merged: &Path) -> IsoResult<()> { let lower = canonical_existing_dir(lower)?; diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index f3d02430e..5ba89c14e 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -45,7 +45,6 @@ rayon.workspace = true regex.workspace = true serde.workspace = true serde_json.workspace = true -similar.workspace = true smallvec.workspace = true syntect.workspace = true tiktoken-rs.workspace = true diff --git a/crates/pi-natives/src/diff.rs b/crates/pi-natives/src/diff.rs new file mode 100644 index 000000000..c357a7c37 --- /dev/null +++ b/crates/pi-natives/src/diff.rs @@ -0,0 +1,1025 @@ +//! jsdiff-compatible diff primitives. +//! +//! # Overview +//! Line, line-array, and word diffs plus a unified-patch hunk builder, all +//! producing byte-identical output to the `diff` npm package (jsdiff v9) under +//! its default options. The Myers O(ND) core is a faithful port of jsdiff's +//! `base.ts`, including its greedy tie-breaking and the edit-graph edge pruning +//! (`minDiagonalToConsider` / `maxDiagonalToConsider`), so change coalescing +//! matches jsdiff run-for-run rather than merely being "a" minimal diff. +//! +//! Everything operates on UTF-16 code units end to end — [`Utf16String`] at +//! the N-API boundary, `&[u16]` internally — which is the exact value space of +//! JS strings. Ill-formed input (unpaired surrogates) is legal content that +//! diffs code-unit-for-code-unit like jsdiff, so callers never need a JS +//! fallback, and no UTF-8 conversion happens in either direction. +//! +//! # Example +//! ```ignore +//! // JS: native.diffLines("a\nb\n", "a\nc\n") +//! // -> [{ value: "a\n", count: 1, added: false, removed: false }, +//! // { value: "b\n", count: 1, added: false, removed: true }, +//! // { value: "c\n", count: 1, added: true, removed: false }] +//! ``` + +use std::{collections::HashMap, rc::Rc}; + +use napi::bindgen_prelude::*; +use napi_derive::napi; + +/// UTF-16 code unit for `\n`. +const LF: u16 = 0x000a; + +/// One jsdiff change object: a run of added, removed, or common tokens. +#[napi(object)] +pub struct DiffChange { + /// Joined token text for this run (lines keep their `\n` terminators). + pub value: Utf16String, + /// Number of tokens in this run. + pub count: u32, + /// True when this run exists only in the new text. + pub added: bool, + /// True when this run exists only in the old text. + pub removed: bool, +} + +/// A change run without its token text, for callers that only need counts. +#[napi(object)] +pub struct DiffRun { + /// Number of tokens in this run. + pub count: u32, + /// True when this run exists only in the new text. + pub added: bool, + /// True when this run exists only in the old text. + pub removed: bool, +} + +/// One hunk of a unified diff, matching jsdiff `structuredPatch` hunks. +#[napi(object)] +pub struct PatchHunk { + /// 1-based first line of the hunk in the old text. + pub old_start: u32, + /// Number of old-text lines covered by the hunk. + pub old_lines: u32, + /// 1-based first line of the hunk in the new text. + pub new_start: u32, + /// Number of new-text lines covered by the hunk. + pub new_lines: u32, + /// Hunk body: `+`/`-`/` `-prefixed lines without trailing newlines, plus + /// `\ No newline at end of file` markers where applicable. + pub lines: Vec, +} + +// ═══════════════════════════════════════════════════════════════════════════ +// Myers core (port of jsdiff base.ts, default options) +// ═══════════════════════════════════════════════════════════════════════════ + +/// A run of tokens sharing one edit classification, in forward order. +#[derive(Clone, Copy)] +struct Run { + count: usize, + added: bool, + removed: bool, +} + +/// Reverse-linked component list node, shared between diagonal paths exactly +/// like jsdiff's `previousComponent` chains (structural sharing keeps the +/// D-path frontier O(D) instead of O(D^2)). +struct Component { + count: usize, + added: bool, + removed: bool, + prev: Option>, +} + +/// Frontier state for one diagonal: furthest old-position reached plus the +/// component chain that got there. +struct PathState { + old_pos: isize, + last: Option>, +} + +/// Extend `path` along its diagonal while tokens match, recording the common +/// run. Returns the new-token position (mirrors jsdiff `extractCommon`). +fn extract_common(path: &mut PathState, new: &[u32], old: &[u32], diagonal: isize) -> isize { + let new_len = new.len() as isize; + let old_len = old.len() as isize; + let mut old_pos = path.old_pos; + let mut new_pos = old_pos - diagonal; + let mut common = 0usize; + while new_pos + 1 < new_len + && old_pos + 1 < old_len + && old[(old_pos + 1) as usize] == new[(new_pos + 1) as usize] + { + new_pos += 1; + old_pos += 1; + common += 1; + } + if common > 0 { + path.last = Some(Rc::new(Component { + count: common, + added: false, + removed: false, + prev: path.last.take(), + })); + } + path.old_pos = old_pos; + new_pos +} + +/// Branch from `path` with one added or removed token (mirrors jsdiff +/// `addToPath`, which merges into the previous component when the edit kind +/// repeats). +fn add_to_path(path: &PathState, added: bool, removed: bool, old_pos_inc: isize) -> PathState { + match &path.last { + Some(last) if last.added == added && last.removed == removed => PathState { + old_pos: path.old_pos + old_pos_inc, + last: Some(Rc::new(Component { + count: last.count + 1, + added, + removed, + prev: last.prev.clone(), + })), + }, + _ => PathState { + old_pos: path.old_pos + old_pos_inc, + last: Some(Rc::new(Component { count: 1, added, removed, prev: path.last.clone() })), + }, + } +} + +/// Convert the winning component chain into forward-ordered runs. +fn build_runs(last: Option>) -> Vec { + let mut runs = Vec::new(); + let mut cursor = last.as_deref(); + while let Some(component) = cursor { + runs.push(Run { + count: component.count, + added: component.added, + removed: component.removed, + }); + cursor = component.prev.as_deref(); + } + runs.reverse(); + runs +} + +/// Myers O(ND) diff over interned token ids, replicating jsdiff's default +/// (non-`oneChangePerToken`, no timeout / `maxEditLength`) execution path so +/// the resulting run structure is identical. +fn myers_diff(old: &[u32], new: &[u32]) -> Vec { + let old_len = old.len() as isize; + let new_len = new.len() as isize; + let max_edit = old_len + new_len; + let offset = max_edit + 1; + let mut best: Vec> = Vec::new(); + best.resize_with((2 * max_edit + 3) as usize, || None); + + // Seed edit length 0: the content may start with common tokens. + let mut seed = PathState { old_pos: -1, last: None }; + let seed_new_pos = extract_common(&mut seed, new, old, 0); + if seed.old_pos + 1 >= old_len && seed_new_pos + 1 >= new_len { + return build_runs(seed.last); + } + best[offset as usize] = Some(seed); + + let mut min_diagonal = isize::MIN; + let mut max_diagonal = isize::MAX; + let mut edit_length: isize = 1; + while edit_length <= max_edit { + let mut diagonal = min_diagonal.max(-edit_length); + while diagonal <= max_diagonal.min(edit_length) { + let idx = (diagonal + offset) as usize; + let remove_path = best[idx - 1].take(); + let add_path_old_pos = best[idx + 1].as_ref().map(|path| path.old_pos); + let can_add = add_path_old_pos.is_some_and(|old_pos| { + let add_new_pos = old_pos - diagonal; + add_new_pos >= 0 && add_new_pos < new_len + }); + let can_remove = remove_path + .as_ref() + .is_some_and(|path| path.old_pos + 1 < old_len); + if !can_add && !can_remove { + best[idx] = None; + diagonal += 2; + continue; + } + + // Branch from the prior path whose old-text position is furthest + // along, preferring the insertion path on ties (jsdiff order). + let mut base_path = if !can_remove + || (can_add + && remove_path.as_ref().is_some_and(|path| { + add_path_old_pos.is_some_and(|add_old| path.old_pos < add_old) + })) { + add_to_path( + best[idx + 1] + .as_ref() + .expect("canAdd implies a live addPath"), + true, + false, + 0, + ) + } else { + add_to_path( + remove_path + .as_ref() + .expect("canRemove implies a live removePath"), + false, + true, + 1, + ) + }; + let new_pos = extract_common(&mut base_path, new, old, diagonal); + if base_path.old_pos + 1 >= old_len && new_pos + 1 >= new_len { + return build_runs(base_path.last); + } + if base_path.old_pos + 1 >= old_len { + max_diagonal = max_diagonal.min(diagonal - 1); + } + if new_pos + 1 >= new_len { + min_diagonal = min_diagonal.max(diagonal + 1); + } + best[idx] = Some(base_path); + diagonal += 2; + } + edit_length += 1; + } + unreachable!("Myers diff terminates within oldLen + newLen edits") +} + +/// Intern each token as a dense id under exact code-unit equality, so the +/// Myers core compares `u32`s instead of re-hashing slices per probe. +fn intern_exact<'a>(old_tokens: &[&'a [u16]], new_tokens: &[&'a [u16]]) -> (Vec, Vec) { + fn assign<'a>(ids: &mut HashMap<&'a [u16], u32>, token: &'a [u16]) -> u32 { + let next = ids.len() as u32; + *ids.entry(token).or_insert(next) + } + let mut ids: HashMap<&'a [u16], u32> = + HashMap::with_capacity(old_tokens.len() + new_tokens.len()); + let old_ids = old_tokens + .iter() + .map(|token| assign(&mut ids, token)) + .collect(); + let new_ids = new_tokens + .iter() + .map(|token| assign(&mut ids, token)) + .collect(); + (old_ids, new_ids) +} + +/// Map runs back to change objects, joining token slices with `join`. +/// Common runs take their text from the new tokens, matching jsdiff +/// `buildValues` with `useLongestToken == false`. +fn build_changes( + runs: &[Run], + old_tokens: &[&[u16]], + new_tokens: &[&[u16]], + join: impl Fn(&[&[u16]]) -> Vec, +) -> Vec { + let mut old_pos = 0usize; + let mut new_pos = 0usize; + runs + .iter() + .map(|run| { + let value = if run.removed { + let value = join(&old_tokens[old_pos..old_pos + run.count]); + old_pos += run.count; + value + } else { + let value = join(&new_tokens[new_pos..new_pos + run.count]); + new_pos += run.count; + if !run.added { + old_pos += run.count; + } + value + }; + DiffChange { + value: value.into(), + count: run.count as u32, + added: run.added, + removed: run.removed, + } + }) + .collect() +} + +// ═══════════════════════════════════════════════════════════════════════════ +// Line diff +// ═══════════════════════════════════════════════════════════════════════════ + +/// jsdiff line tokenization under default options: each token is a line +/// including its `\n` (or `\r\n`) terminator; a final line without a newline +/// is kept as-is; a lone `\r` never terminates a line. +fn line_tokens(text: &[u16]) -> Vec<&[u16]> { + text.split_inclusive(|&unit| unit == LF).collect() +} + +fn diff_line_tokens(old_tokens: &[&[u16]], new_tokens: &[&[u16]]) -> Vec { + let (old_ids, new_ids) = intern_exact(old_tokens, new_tokens); + myers_diff(&old_ids, &new_ids) +} + +/// Concatenate token slices (jsdiff line `join`). +fn concat_tokens(tokens: &[&[u16]]) -> Vec { + let mut out = Vec::with_capacity(tokens.iter().map(|token| token.len()).sum()); + for token in tokens { + out.extend_from_slice(token); + } + out +} + +/// Line diff with jsdiff `diffLines(oldText, newText)` semantics (default +/// options). Change values keep line terminators, and common runs are joined +/// from the new text. +#[napi] +pub fn diff_lines(old_text: Utf16String, new_text: Utf16String) -> Vec { + diff_lines_impl(&old_text, &new_text) +} + +fn diff_lines_impl(old_text: &[u16], new_text: &[u16]) -> Vec { + let old_tokens = line_tokens(old_text); + let new_tokens = line_tokens(new_text); + let runs = diff_line_tokens(&old_tokens, &new_tokens); + build_changes(&runs, &old_tokens, &new_tokens, concat_tokens) +} + +/// Diff `oldText.split("\n")` against `newText.split("\n")` with jsdiff +/// `diffArrays` semantics (exact code-unit equality, empty lines preserved), +/// returning only run lengths. +/// +/// Callers that map line numbers — like hashline recovery — need the counts, +/// not another copy of the text. +#[napi] +pub fn diff_line_runs(old_text: Utf16String, new_text: Utf16String) -> Vec { + diff_line_runs_impl(&old_text, &new_text) +} + +fn diff_line_runs_impl(old_text: &[u16], new_text: &[u16]) -> Vec { + let old_tokens: Vec<&[u16]> = old_text.split(|&unit| unit == LF).collect(); + let new_tokens: Vec<&[u16]> = new_text.split(|&unit| unit == LF).collect(); + let (old_ids, new_ids) = intern_exact(&old_tokens, &new_tokens); + myers_diff(&old_ids, &new_ids) + .into_iter() + .map(|run| DiffRun { count: run.count as u32, added: run.added, removed: run.removed }) + .collect() +} + +// ═══════════════════════════════════════════════════════════════════════════ +// Structured patch (port of jsdiff patch/create.ts hunk builder) +// ═══════════════════════════════════════════════════════════════════════════ + +/// Prepend a `+`/`-`/` ` marker to a line's code units. +fn prefixed_line(prefix: u8, line: &[u16]) -> Vec { + let mut out = Vec::with_capacity(1 + line.len()); + out.push(u16::from(prefix)); + out.extend_from_slice(line); + out +} + +/// `\ No newline at end of file`, as UTF-16 code units. +fn no_newline_marker() -> Vec { + "\\ No newline at end of file".encode_utf16().collect() +} + +/// Unified-diff hunks with jsdiff +/// `structuredPatch(_, _, oldText, newText, _, _, { context }).hunks` +/// semantics. `context` defaults to 4 like jsdiff. +#[napi] +pub fn structured_patch_hunks( + old_text: Utf16String, + new_text: Utf16String, + context: Option, +) -> Vec { + structured_patch_hunks_impl(&old_text, &new_text, context) +} + +fn structured_patch_hunks_impl( + old_text: &[u16], + new_text: &[u16], + context: Option, +) -> Vec { + let context = context.map_or(4usize, |value| value as usize); + let old_tokens = line_tokens(old_text); + let new_tokens = line_tokens(new_text); + let runs = diff_line_tokens(&old_tokens, &new_tokens); + + // Change list with per-change line slices; the trailing sentinel mirrors + // jsdiff's pushed empty change that flushes the final hunk. + struct ChangeLines<'a> { + added: bool, + removed: bool, + lines: &'a [&'a [u16]], + } + let mut list: Vec = Vec::with_capacity(runs.len() + 1); + let mut old_pos = 0usize; + let mut new_pos = 0usize; + for run in &runs { + let lines: &[&[u16]] = if run.removed { + let slice = &old_tokens[old_pos..old_pos + run.count]; + old_pos += run.count; + slice + } else { + let slice = &new_tokens[new_pos..new_pos + run.count]; + new_pos += run.count; + if !run.added { + old_pos += run.count; + } + slice + }; + list.push(ChangeLines { added: run.added, removed: run.removed, lines }); + } + list.push(ChangeLines { added: false, removed: false, lines: &[] }); + + // Hunk skeleton before the trailing-newline post-pass; lines stay `Vec` + // so the pass below can pop terminators in place. + struct RawHunk { + old_start: usize, + old_lines: usize, + new_start: usize, + new_lines: usize, + lines: Vec>, + } + let mut hunks: Vec = Vec::new(); + let mut old_range_start = 0usize; + let mut new_range_start = 0usize; + let mut cur_range: Vec> = Vec::new(); + let mut old_line = 1usize; + let mut new_line = 1usize; + for i in 0..list.len() { + let current = &list[i]; + if current.added || current.removed { + // Open a hunk seeded with trailing context from the previous + // common run. + if old_range_start == 0 { + old_range_start = old_line; + new_range_start = new_line; + if i > 0 && context > 0 { + let prev_lines = list[i - 1].lines; + let take = prev_lines.len().min(context); + cur_range = prev_lines[prev_lines.len() - take..] + .iter() + .map(|line| prefixed_line(b' ', line)) + .collect(); + old_range_start -= cur_range.len(); + new_range_start -= cur_range.len(); + } + } + let marker = if current.added { b'+' } else { b'-' }; + for line in current.lines { + cur_range.push(prefixed_line(marker, line)); + } + if current.added { + new_line += current.lines.len(); + } else { + old_line += current.lines.len(); + } + } else { + if old_range_start != 0 { + if current.lines.len() <= context * 2 && i + 2 < list.len() { + // Common run small enough to join adjacent hunks. + for line in current.lines { + cur_range.push(prefixed_line(b' ', line)); + } + } else { + // Close the hunk with leading context. + let context_size = current.lines.len().min(context); + for line in ¤t.lines[..context_size] { + cur_range.push(prefixed_line(b' ', line)); + } + hunks.push(RawHunk { + old_start: old_range_start, + old_lines: old_line - old_range_start + context_size, + new_start: new_range_start, + new_lines: new_line - new_range_start + context_size, + lines: std::mem::take(&mut cur_range), + }); + old_range_start = 0; + new_range_start = 0; + } + } + old_line += current.lines.len(); + new_line += current.lines.len(); + } + } + + // Strip trailing newlines and add "no newline at EOF" markers. + for hunk in &mut hunks { + let mut i = 0; + while i < hunk.lines.len() { + if hunk.lines[i].last() == Some(&LF) { + hunk.lines[i].pop(); + } else { + hunk.lines.insert(i + 1, no_newline_marker()); + i += 1; + } + i += 1; + } + } + hunks + .into_iter() + .map(|hunk| PatchHunk { + old_start: hunk.old_start as u32, + old_lines: hunk.old_lines as u32, + new_start: hunk.new_start as u32, + new_lines: hunk.new_lines as u32, + lines: hunk.lines.into_iter().map(Utf16String::from).collect(), + }) + .collect() +} + +// ═══════════════════════════════════════════════════════════════════════════ +// Word diff (port of jsdiff word.ts, default options) +// ═══════════════════════════════════════════════════════════════════════════ + +/// jsdiff's `extendedWordChars` class: Latin-script word characters. Takes a +/// code point so astral input classifies like a JS regex with the `u` flag — +/// never a word character (every member is BMP). +const fn is_word_char(cp: u32) -> bool { + matches!(cp, + 0x30..=0x39 // 0-9 + | 0x41..=0x5A // A-Z + | 0x5F // _ + | 0x61..=0x7A // a-z + | 0xAD + | 0xC0..=0xD6 + | 0xD8..=0xF6 + | 0xF8..=0x2C6 + | 0x2C8..=0x2D7 + | 0x2DE..=0x2FF + | 0x1E00..=0x1EFF) +} + +/// JavaScript's `\s` / `String.prototype.trim` whitespace set (`WhiteSpace` + +/// `LineTerminator` productions). Every member is a single UTF-16 code unit, +/// so unit-level scans here match jsdiff's code-unit-level scans exactly. +const fn is_js_whitespace(cp: u32) -> bool { + matches!( + cp, + 0x09 | 0x0a | 0x0b | 0x0c | 0x0d | 0x20 | 0xa0 | 0x1680 | 0x2000 + ..=0x200a | 0x2028 | 0x2029 | 0x202f | 0x205f | 0x3000 | 0xfeff + ) +} + +const fn is_ws_unit(unit: u16) -> bool { + is_js_whitespace(unit as u32) +} + +fn trim_leading_ws(s: &[u16]) -> &[u16] { + let start = s + .iter() + .position(|&unit| !is_ws_unit(unit)) + .unwrap_or(s.len()); + &s[start..] +} + +fn trim_trailing_ws(s: &[u16]) -> &[u16] { + let end = s + .iter() + .rposition(|&unit| !is_ws_unit(unit)) + .map_or(0, |i| i + 1); + &s[..end] +} + +fn leading_ws(s: &[u16]) -> &[u16] { + &s[..s.len() - trim_leading_ws(s).len()] +} + +fn trailing_ws(s: &[u16]) -> &[u16] { + &s[trim_trailing_ws(s).len()..] +} + +fn js_trim(s: &[u16]) -> &[u16] { + trim_trailing_ws(trim_leading_ws(s)) +} + +/// Iterator over `(start, code_point, unit_len)` that pairs surrogates and +/// passes unpaired surrogates through as their own code points, exactly like +/// JS regex scanning under the `u` flag. +struct CodePoints<'a> { + text: &'a [u16], + pos: usize, +} + +impl Iterator for CodePoints<'_> { + type Item = (usize, u32, usize); + + fn next(&mut self) -> Option { + let &unit = self.text.get(self.pos)?; + let start = self.pos; + if matches!(unit, 0xd800..=0xdbff) + && let Some(&low) = self.text.get(start + 1) + && matches!(low, 0xdc00..=0xdfff) + { + self.pos += 2; + let cp = 0x10000 + ((u32::from(unit & 0x3ff) << 10) | u32::from(low & 0x3ff)); + return Some((start, cp, 2)); + } + self.pos += 1; + Some((start, u32::from(unit), 1)) + } +} + +const fn code_points(text: &[u16]) -> CodePoints<'_> { + CodePoints { text, pos: 0 } +} + +/// Raw regex-equivalent scan: word runs, whitespace runs, or single other +/// code points (jsdiff `tokenizeIncludingWhitespace` with the `u` flag). +fn word_parts(text: &[u16]) -> Vec<&[u16]> { + let mut parts = Vec::new(); + let mut iter = code_points(text).peekable(); + while let Some((start, cp, len)) = iter.next() { + let class = if is_word_char(cp) { + 1u8 + } else if is_js_whitespace(cp) { + 2u8 + } else { + 0u8 + }; + let mut end = start + len; + if class != 0 { + while let Some(&(_, next_cp, next_len)) = iter.peek() { + let same = if class == 1 { + is_word_char(next_cp) + } else { + is_js_whitespace(next_cp) + }; + if !same { + break; + } + end += next_len; + iter.next(); + } + } + parts.push(&text[start..end]); + } + parts +} + +/// jsdiff `WordDiff.tokenize`: stitch whitespace runs onto adjacent word or +/// punctuation parts, duplicating interior whitespace into both neighbors. +fn word_tokens(text: &[u16]) -> Vec> { + let parts = word_parts(text); + let mut tokens: Vec> = Vec::with_capacity(parts.len()); + let mut prev_part: Option<&[u16]> = None; + for part in parts { + let part_is_ws = part.first().is_some_and(|&unit| is_ws_unit(unit)); + if part_is_ws { + if prev_part.is_none() { + tokens.push(part.to_vec()); + } else { + let last = tokens + .last_mut() + .expect("tokens non-empty after first part"); + last.extend_from_slice(part); + } + } else if let Some(prev) = + prev_part.filter(|p| p.first().is_some_and(|&unit| is_ws_unit(unit))) + { + if tokens.last().is_some_and(|last| last.as_slice() == prev) { + let last = tokens.last_mut().expect("checked non-empty"); + last.extend_from_slice(part); + } else { + let mut token = Vec::with_capacity(prev.len() + part.len()); + token.extend_from_slice(prev); + token.extend_from_slice(part); + tokens.push(token); + } + } else { + tokens.push(part.to_vec()); + } + prev_part = Some(part); + } + tokens +} + +/// jsdiff `WordDiff.join`: concatenate, stripping leading whitespace from +/// every token after the first. +fn word_join(tokens: &[&[u16]]) -> Vec { + let mut out = Vec::new(); + for (i, token) in tokens.iter().enumerate() { + if i == 0 { + out.extend_from_slice(token); + } else { + out.extend_from_slice(trim_leading_ws(token)); + } + } + out +} + +fn longest_common_prefix<'a>(a: &'a [u16], b: &[u16]) -> &'a [u16] { + let len = a.iter().zip(b).take_while(|(x, y)| x == y).count(); + &a[..len] +} + +fn longest_common_suffix<'a>(a: &'a [u16], b: &[u16]) -> &'a [u16] { + let len = a + .iter() + .rev() + .zip(b.iter().rev()) + .take_while(|(x, y)| x == y) + .count(); + &a[a.len() - len..] +} + +fn remove_prefix(s: &[u16], prefix: &[u16]) -> Vec { + s.strip_prefix(prefix) + .expect("value must start with recorded prefix") + .to_vec() +} + +fn remove_suffix(s: &[u16], suffix: &[u16]) -> Vec { + s.strip_suffix(suffix) + .expect("value must end with recorded suffix") + .to_vec() +} + +fn replace_prefix(s: &[u16], old_prefix: &[u16], new_prefix: &[u16]) -> Vec { + let rest = s + .strip_prefix(old_prefix) + .expect("value must start with recorded prefix"); + let mut out = Vec::with_capacity(new_prefix.len() + rest.len()); + out.extend_from_slice(new_prefix); + out.extend_from_slice(rest); + out +} + +fn replace_suffix(s: &[u16], old_suffix: &[u16], new_suffix: &[u16]) -> Vec { + let rest = s + .strip_suffix(old_suffix) + .expect("value must end with recorded suffix"); + let mut out = Vec::with_capacity(rest.len() + new_suffix.len()); + out.extend_from_slice(rest); + out.extend_from_slice(new_suffix); + out +} + +/// jsdiff `maximumOverlap`: the longest prefix of `b` that is also a suffix +/// of `a`, via the KMP failure function over code units. +fn maximum_overlap<'a>(a: &[u16], b: &'a [u16]) -> &'a [u16] { + let start_a = a.len().saturating_sub(b.len()); + let end_b = b.len().min(a.len()); + if end_b == 0 { + return &[]; + } + let mut map = vec![0usize; end_b]; + let mut k = 0usize; + for j in 1..end_b { + if b[j] == b[k] { + map[j] = map[k]; + } else { + map[j] = k; + } + while k > 0 && b[j] != b[k] { + k = map[k]; + } + if b[j] == b[k] { + k += 1; + } + } + k = 0; + for &unit in &a[start_a..] { + while k > 0 && unit != b[k] { + k = map[k]; + } + if unit == b[k] { + k += 1; + } + } + &b[..k] +} + +/// jsdiff `dedupeWhitespaceInChangeObjects` (no segmenter): trim whitespace +/// that the tokenizer duplicated across a keep/delete/insert boundary. +fn dedupe_whitespace( + changes: &mut [DiffChange], + start_keep: Option, + deletion: Option, + insertion: Option, + end_keep: Option, +) { + match (deletion, insertion) { + (Some(del), Some(ins)) => { + let old_ws_prefix = leading_ws(&changes[del].value).to_vec(); + let old_ws_suffix = trailing_ws(&changes[del].value).to_vec(); + let new_ws_prefix = leading_ws(&changes[ins].value).to_vec(); + let new_ws_suffix = trailing_ws(&changes[ins].value).to_vec(); + if let Some(start) = start_keep { + let common_ws_prefix = longest_common_prefix(&old_ws_prefix, &new_ws_prefix).to_vec(); + changes[start].value = + replace_suffix(&changes[start].value, &new_ws_prefix, &common_ws_prefix).into(); + changes[del].value = remove_prefix(&changes[del].value, &common_ws_prefix).into(); + changes[ins].value = remove_prefix(&changes[ins].value, &common_ws_prefix).into(); + } + if let Some(end) = end_keep { + let common_ws_suffix = longest_common_suffix(&old_ws_suffix, &new_ws_suffix).to_vec(); + changes[end].value = + replace_prefix(&changes[end].value, &new_ws_suffix, &common_ws_suffix).into(); + changes[del].value = remove_suffix(&changes[del].value, &common_ws_suffix).into(); + changes[ins].value = remove_suffix(&changes[ins].value, &common_ws_suffix).into(); + } + }, + (None, Some(ins)) => { + if start_keep.is_some() { + let ws_len = leading_ws(&changes[ins].value).len(); + changes[ins].value = changes[ins].value[ws_len..].to_vec().into(); + } + if let Some(end) = end_keep { + let ws_len = leading_ws(&changes[end].value).len(); + changes[end].value = changes[end].value[ws_len..].to_vec().into(); + } + }, + (Some(del), None) => match (start_keep, end_keep) { + (Some(start), Some(end)) => { + let new_ws_full = leading_ws(&changes[end].value).to_vec(); + let del_ws_start = leading_ws(&changes[del].value).to_vec(); + let del_ws_end = trailing_ws(&changes[del].value).to_vec(); + let new_ws_start = longest_common_prefix(&new_ws_full, &del_ws_start).to_vec(); + changes[del].value = remove_prefix(&changes[del].value, &new_ws_start).into(); + let new_ws_end = + longest_common_suffix(&new_ws_full[new_ws_start.len()..], &del_ws_end).to_vec(); + changes[del].value = remove_suffix(&changes[del].value, &new_ws_end).into(); + changes[end].value = + replace_prefix(&changes[end].value, &new_ws_full, &new_ws_end).into(); + let start_ws = &new_ws_full[..new_ws_full.len() - new_ws_end.len()]; + changes[start].value = + replace_suffix(&changes[start].value, &new_ws_full, start_ws).into(); + }, + (None, Some(end)) => { + let end_keep_ws_prefix = leading_ws(&changes[end].value).to_vec(); + let deletion_ws_suffix = trailing_ws(&changes[del].value).to_vec(); + let overlap = maximum_overlap(&deletion_ws_suffix, &end_keep_ws_prefix).to_vec(); + changes[del].value = remove_suffix(&changes[del].value, &overlap).into(); + }, + (Some(start), None) => { + let start_keep_ws_suffix = trailing_ws(&changes[start].value).to_vec(); + let deletion_ws_prefix = leading_ws(&changes[del].value).to_vec(); + let overlap = maximum_overlap(&start_keep_ws_suffix, &deletion_ws_prefix).to_vec(); + changes[del].value = remove_prefix(&changes[del].value, &overlap).into(); + }, + (None, None) => {}, + }, + (None, None) => {}, + } +} + +/// jsdiff `WordDiff.postProcess` under default options. +fn word_post_process(changes: &mut [DiffChange]) { + let mut last_keep: Option = None; + let mut insertion: Option = None; + let mut deletion: Option = None; + for i in 0..changes.len() { + if changes[i].added { + insertion = Some(i); + } else if changes[i].removed { + deletion = Some(i); + } else { + if insertion.is_some() || deletion.is_some() { + dedupe_whitespace(changes, last_keep, deletion, insertion, Some(i)); + } + last_keep = Some(i); + insertion = None; + deletion = None; + } + } + if insertion.is_some() || deletion.is_some() { + dedupe_whitespace(changes, last_keep, deletion, insertion, None); + } +} + +/// Word diff with jsdiff `diffWords(oldText, newText)` semantics (default +/// options). +/// +/// Tokens carry surrounding whitespace, equality ignores it, and the +/// post-pass dedupes whitespace across change boundaries. +#[napi] +pub fn diff_words(old_text: Utf16String, new_text: Utf16String) -> Vec { + diff_words_impl(&old_text, &new_text) +} + +fn diff_words_impl(old_text: &[u16], new_text: &[u16]) -> Vec { + let old_tokens = word_tokens(old_text); + let new_tokens = word_tokens(new_text); + let old_refs: Vec<&[u16]> = old_tokens.iter().map(Vec::as_slice).collect(); + let new_refs: Vec<&[u16]> = new_tokens.iter().map(Vec::as_slice).collect(); + // Equality is whitespace-insensitive: intern by trimmed text. + let old_keys: Vec<&[u16]> = old_refs.iter().map(|token| js_trim(token)).collect(); + let new_keys: Vec<&[u16]> = new_refs.iter().map(|token| js_trim(token)).collect(); + let (old_ids, new_ids) = intern_exact(&old_keys, &new_keys); + let runs = myers_diff(&old_ids, &new_ids); + let mut changes = build_changes(&runs, &old_refs, &new_refs, word_join); + word_post_process(&mut changes); + changes +} + +#[cfg(test)] +mod tests { + use super::*; + + fn u16s(text: &str) -> Vec { + text.encode_utf16().collect() + } + + fn lines(old: &str, new: &str) -> Vec<(String, bool, bool)> { + diff_lines_impl(&u16s(old), &u16s(new)) + .into_iter() + .map(|c| (String::from_utf16(&c.value).unwrap(), c.added, c.removed)) + .collect() + } + + #[test] + fn line_diff_replaces_middle_line() { + assert_eq!(lines("a\nb\nc\n", "a\nx\nc\n"), vec![ + ("a\n".into(), false, false), + ("b\n".into(), false, true), + ("x\n".into(), true, false), + ("c\n".into(), false, false), + ]); + } + + #[test] + fn line_diff_treats_missing_trailing_newline_as_distinct() { + assert_eq!(lines("a\nb", "a\nb\n"), vec![ + ("a\n".into(), false, false), + ("b".into(), false, true), + ("b\n".into(), true, false), + ]); + } + + #[test] + fn structured_patch_marks_missing_eof_newline() { + let hunks = structured_patch_hunks_impl(&u16s("a\nb"), &u16s("a\nc"), Some(3)); + assert_eq!(hunks.len(), 1); + let body: Vec = hunks[0] + .lines + .iter() + .map(|line| String::from_utf16(line).unwrap()) + .collect(); + assert_eq!(body, vec![ + " a", + "-b", + "\\ No newline at end of file", + "+c", + "\\ No newline at end of file" + ]); + } + + #[test] + fn word_diff_dedupes_boundary_whitespace() { + // jsdiff's documented example 2: K:'foo ' D:'bar' I:'qux' K:' baz'. + let changes = diff_words_impl(&u16s("foo bar baz"), &u16s("foo qux baz")); + let shaped: Vec<(String, bool, bool)> = changes + .into_iter() + .map(|c| (String::from_utf16(&c.value).unwrap(), c.added, c.removed)) + .collect(); + assert_eq!(shaped, vec![ + ("foo ".into(), false, false), + ("bar".into(), false, true), + ("qux".into(), true, false), + (" baz".into(), false, false), + ]); + } + + #[test] + fn line_runs_preserve_empty_lines() { + let runs = diff_line_runs_impl(&u16s("a\n\nb"), &u16s("a\n\nc")); + let shaped: Vec<(u32, bool, bool)> = runs + .into_iter() + .map(|r| (r.count, r.added, r.removed)) + .collect(); + assert_eq!(shaped, vec![(2, false, false), (1, false, true), (1, true, false)]); + } + + #[test] + fn unpaired_surrogates_diff_as_distinct_content() { + // Lone surrogates are legal JS string content; they must compare by + // code unit instead of failing (or lossily surviving) a UTF-8 round + // trip. + let old = [0x61, 0xd800, LF]; + let new = [0x61, 0xd801, LF]; + let shaped: Vec<(Vec, bool, bool)> = diff_lines_impl(&old, &new) + .into_iter() + .map(|c| (c.value.to_vec(), c.added, c.removed)) + .collect(); + assert_eq!(shaped, vec![(old.to_vec(), false, true), (new.to_vec(), true, false)]); + } + + #[test] + fn word_scan_keeps_lone_surrogate_before_astral_pair_separate() { + // "\u{D800}🚀" is a lone high surrogate directly followed by a valid + // pair; the `u`-flag scan must yield two "other" tokens, so replacing + // only the rocket leaves the lone surrogate as common content. + let old: Vec = [0xd800, 0xd83d, 0xde80].to_vec(); // "\u{D800}🚀" + let new: Vec = [0xd800, 0x78].to_vec(); // "\u{D800}x" + let shaped: Vec<(Vec, bool, bool)> = diff_words_impl(&old, &new) + .into_iter() + .map(|c| (c.value.to_vec(), c.added, c.removed)) + .collect(); + assert_eq!(shaped, vec![ + (vec![0xd800], false, false), + (vec![0xd83d, 0xde80], false, true), + (vec![0x78], true, false), + ]); + } +} diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 7611a3654..162e11bd1 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -27,6 +27,7 @@ pub mod ast; pub mod block; pub mod clipboard; pub mod crash_handler; +pub mod diff; pub mod fd; pub mod glob; pub mod glob_util; @@ -53,6 +54,7 @@ pub(crate) mod testing; pub mod text; pub mod tokens; pub(crate) mod utils; +pub mod vectors; pub mod workspace; #[cfg(target_os = "windows")] @@ -248,7 +250,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV17_0_0")] +#[napi(js_name = "__piNativesV17_0_8")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/crates/pi-natives/src/pty.rs b/crates/pi-natives/src/pty.rs index f07912909..bfdb3cb65 100644 --- a/crates/pi-natives/src/pty.rs +++ b/crates/pi-natives/src/pty.rs @@ -132,7 +132,8 @@ impl PtySession { Self { core: Arc::new(Mutex::new(None)) } } - /// Start a shell command and stream output chunks via callback. + /// Start a shell command, stream output chunks, and report the spawned child + /// PID. #[napi] pub fn start<'env>( &self, @@ -140,6 +141,8 @@ impl PtySession { options: PtyStartOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] on_chunk: Option>, + #[napi(ts_arg_type = "((error: Error | null, pid: number) => void) | undefined | null")] + on_start: Option>, ) -> Result> { let run_config = PtyRunConfig { command: PtyCommand::Shell { command: options.command, shell: options.shell }, @@ -148,11 +151,11 @@ impl PtySession { cols: options.cols.unwrap_or(120).clamp(20, 400), rows: options.rows.unwrap_or(40).clamp(5, 200), }; - self.start_config(env, run_config, options.timeout_ms, options.signal, on_chunk) + self.start_config(env, run_config, options.timeout_ms, options.signal, on_chunk, on_start) } - /// Start an executable with separate arguments and stream output chunks via - /// callback. + /// Start an executable with separate arguments, stream output chunks, and + /// report the spawned child PID. #[napi] pub fn start_argv<'env>( &self, @@ -160,6 +163,8 @@ impl PtySession { options: PtyArgvStartOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] on_chunk: Option>, + #[napi(ts_arg_type = "((error: Error | null, pid: number) => void) | undefined | null")] + on_start: Option>, ) -> Result> { let run_config = PtyRunConfig { command: PtyCommand::Argv { application: options.application, args: options.args }, @@ -168,7 +173,7 @@ impl PtySession { cols: options.cols.unwrap_or(120).clamp(20, 400), rows: options.rows.unwrap_or(40).clamp(5, 200), }; - self.start_config(env, run_config, options.timeout_ms, options.signal, on_chunk) + self.start_config(env, run_config, options.timeout_ms, options.signal, on_chunk, on_start) } /// Write raw input bytes to PTY stdin. @@ -201,6 +206,7 @@ impl PtySession { timeout_ms: Option, signal: Option>, on_chunk: Option>, + on_start: Option>, ) -> Result> { let ct = task::CancelToken::new(timeout_ms, signal); let core = Arc::clone(&self.core); @@ -215,9 +221,10 @@ impl PtySession { *guard = Some(PtySessionCore { control_tx }); } task::future(env, "pty.start", async move { - let run_result = - tokio::task::spawn_blocking(move || run_pty_sync(run_config, on_chunk, control_rx, ct)) - .await; + let run_result = tokio::task::spawn_blocking(move || { + run_pty_sync(run_config, on_chunk, on_start, control_rx, ct) + }) + .await; let mut guard = core.lock(); *guard = None; @@ -262,6 +269,7 @@ fn terminate_pty_processes( fn run_pty_sync( config: PtyRunConfig, on_chunk: Option>, + on_start: Option>, control_rx: flume::Receiver, ct: task::CancelToken, ) -> Result { @@ -343,6 +351,11 @@ fn run_pty_sync( .spawn_command(cmd) .map_err(|err| Error::from_reason(format!("Failed to spawn PTY command: {err}")))?; drop(pair.slave); + let child_process_id = child.process_id(); + let child_pid = child_process_id.and_then(|value| i32::try_from(value).ok()); + if let Some(callback) = on_start.as_ref() { + callback.call(Ok(child_process_id.unwrap_or(0)), ThreadsafeFunctionCallMode::NonBlocking); + } ct.heartbeat() .map_err(|err| Error::from_reason(format!("PTY setup cancelled before reader: {err}")))?; @@ -423,9 +436,6 @@ fn run_pty_sync( let _ = reader_tx.send(ReaderEvent::Done); }); - let child_pid = child - .process_id() - .and_then(|value| i32::try_from(value).ok()); #[cfg(unix)] let process_group_id = master.process_group_leader().filter(|pgid| *pgid > 0); #[cfg(not(unix))] diff --git a/crates/pi-natives/src/shell.rs b/crates/pi-natives/src/shell.rs index ccc0e07dd..c2aea1d13 100644 --- a/crates/pi-natives/src/shell.rs +++ b/crates/pi-natives/src/shell.rs @@ -558,4 +558,38 @@ mod tests { .expect("shell run should return"); assert!(result.cancelled); } + + #[tokio::test(flavor = "multi_thread")] + async fn timeout_drains_pipeline_output_before_stopping_reader() { + let shell = CoreShell::new(None); + let (tx, rx) = flume::unbounded::(); + // `tail` runs as an in-process builtin, so cancellation kills only the + // external `yes`; tail then sees EOF and flushes its final 5 lines into + // the post-cancel reader grace window. The deadline must be generous + // enough that `yes` has demonstrably spawned and produced before the + // timeout fires — a 50ms budget lost that race on cold CI runners and + // tail flushed an empty ring buffer. + const TIMEOUT_MS: u32 = 750; + let result = shell + .run( + CoreShellRunOptions { + command: "yes x | tail -5".to_string(), + cwd: None, + env: None, + timeout_ms: Some(TIMEOUT_MS), + }, + Some(tx), + CancelToken::new(Some(TIMEOUT_MS)), + ) + .await + .expect("shell run"); + + let mut output = String::new(); + while let Ok(chunk) = rx.recv_async().await { + output.push_str(&chunk); + } + + assert!(result.timed_out); + assert_eq!(output.lines().filter(|line| *line == "x").count(), 5); + } } diff --git a/crates/pi-natives/src/text.rs b/crates/pi-natives/src/text.rs index 56de90256..25da9cab1 100644 --- a/crates/pi-natives/src/text.rs +++ b/crates/pi-natives/src/text.rs @@ -23,6 +23,7 @@ const MIN_TAB_WIDTH: u32 = 1; const MAX_TAB_WIDTH: u32 = 16; pub const DEFAULT_TAB_WIDTH: usize = 3; const ESC: u16 = 0x1b; +const OSC8_CLOSE: [u16; 6] = [ESC, b']' as u16, b'8' as u16, b';' as u16, b';' as u16, 0x07]; #[inline] fn clamp_tab_width_for_ops(width: u32) -> usize { @@ -237,6 +238,42 @@ impl AnsiState { } } +#[derive(Default)] +struct WrapState { + sgr: AnsiState, + hyperlink: Option>, +} + +impl WrapState { + #[inline] + const fn new() -> Self { + Self { sgr: AnsiState::new(), hyperlink: None } + } + + #[inline] + fn apply_ansi_u16(&mut self, seq: &[u16]) { + if is_sgr_u16(seq) { + self.sgr.apply_sgr_u16(&seq[2..seq.len() - 1]); + } else if let Some(uri) = osc8_uri_u16(seq) { + if uri.is_empty() { + self.hyperlink = None; + } else { + let hyperlink = self.hyperlink.get_or_insert_default(); + hyperlink.clear(); + hyperlink.extend_from_slice(seq); + } + } + } + + #[inline] + fn write_restore_u16(&self, out: &mut Vec) { + self.sgr.write_restore_u16(out); + if let Some(hyperlink) = &self.hyperlink { + out.extend_from_slice(hyperlink); + } + } +} + #[inline] fn write_color_u16(out: &mut Vec, color: ColorVal, base: u32, first: &mut bool) { if color == COLOR_NONE { @@ -372,6 +409,28 @@ fn is_sgr_u16(seq: &[u16]) -> bool { seq.len() >= 3 && seq[1] == b'[' as u16 && *seq.last().unwrap() == b'm' as u16 } +#[inline] +fn osc8_uri_u16(seq: &[u16]) -> Option<&[u16]> { + if seq.len() < OSC8_CLOSE.len() + || seq[0] != ESC + || seq[1] != b']' as u16 + || seq[2] != b'8' as u16 + || seq[3] != b';' as u16 + { + return None; + } + + let body_end = if seq.last() == Some(&0x07_u16) { + seq.len() - 1 + } else if seq.ends_with(&[ESC, b'\\' as u16]) { + seq.len() - 2 + } else { + return None; + }; + let uri_start = seq[4..body_end].iter().position(|&u| u == b';' as u16)? + 5; + Some(&seq[uri_start..body_end]) +} + struct Osc66Info<'a> { payload: &'a [u16], scale: usize, @@ -852,43 +911,44 @@ fn flush_pending_ansi( // ============================================================================ #[inline] -fn write_active_codes(state: &AnsiState, out: &mut Vec) { - if !state.is_empty() { - state.write_restore_u16(out); +fn write_active_codes(state: &WrapState, out: &mut Vec) { + state.write_restore_u16(out); +} + +#[inline] +fn write_hyperlink_close(state: &WrapState, out: &mut Vec) { + if state.hyperlink.is_some() { + out.extend_from_slice(&OSC8_CLOSE); } } #[inline] -fn write_line_end_reset(state: &AnsiState, out: &mut Vec) { - let has_underline = state.attrs & ATTR_UNDERLINE != 0; - let has_strike = state.attrs & ATTR_STRIKE != 0; - if !has_underline && !has_strike { - return; - } - - out.extend_from_slice(&[ESC, b'[' as u16]); - if has_underline { - out.extend_from_slice(&[b'2' as u16, b'4' as u16]); - if has_strike { - out.push(b';' as u16); +fn write_line_end_reset(state: &WrapState, out: &mut Vec) { + let has_underline = state.sgr.attrs & ATTR_UNDERLINE != 0; + let has_strike = state.sgr.attrs & ATTR_STRIKE != 0; + if has_underline || has_strike { + out.extend_from_slice(&[ESC, b'[' as u16]); + if has_underline { + out.extend_from_slice(&[b'2' as u16, b'4' as u16]); + if has_strike { + out.push(b';' as u16); + } } + if has_strike { + out.extend_from_slice(&[b'2' as u16, b'9' as u16]); + } + out.push(b'm' as u16); } - if has_strike { - out.extend_from_slice(&[b'2' as u16, b'9' as u16]); - } - out.push(b'm' as u16); + write_hyperlink_close(state, out); } -fn update_state_from_text(data: &[u16], state: &mut AnsiState) { +fn update_state_from_text(data: &[u16], state: &mut WrapState) { let mut i = 0usize; while i < data.len() { if data[i] == ESC && let Some(seq_len) = ansi_seq_len_u16(data, i) { - let seq = &data[i..i + seq_len]; - if is_sgr_u16(seq) { - state.apply_sgr_u16(&seq[2..seq_len - 1]); - } + state.apply_ansi_u16(&data[i..i + seq_len]); i += seq_len; continue; } @@ -977,7 +1037,7 @@ fn break_long_word( word: &[u16], width: usize, tab_width: usize, - state: &mut AnsiState, + state: &mut WrapState, ) -> SmallVec<[Vec; 4]> { let mut lines = SmallVec::<[Vec; 4]>::new(); let mut current_line = Vec::::new(); @@ -1004,9 +1064,7 @@ fn break_long_word( continue; } current_line.extend_from_slice(seq); - if is_sgr_u16(seq) { - state.apply_sgr_u16(&seq[2..seq_len - 1]); - } + state.apply_ansi_u16(seq); i += seq_len; continue; } @@ -1070,14 +1128,18 @@ fn wrap_single_line(line: &[u16], width: usize, tab_width: usize) -> SmallVec<[V } if visible_width_u16(line, tab_width) <= width { - return smallvec![line.to_vec()]; + let mut only = line.to_vec(); + let mut state = WrapState::new(); + update_state_from_text(line, &mut state); + write_hyperlink_close(&state, &mut only); + return smallvec![only]; } let tokens = split_into_tokens_with_ansi(line); let mut wrapped = SmallVec::<[Vec; 4]>::new(); let mut current_line = Vec::::new(); let mut current_width = 0usize; - let mut state = AnsiState::new(); + let mut state = WrapState::new(); for token in tokens { let token_width = visible_width_u16(&token, tab_width); @@ -1124,6 +1186,7 @@ fn wrap_single_line(line: &[u16], width: usize, tab_width: usize) -> SmallVec<[V } if !current_line.is_empty() { + write_hyperlink_close(&state, &mut current_line); wrapped.push(current_line); } @@ -1148,7 +1211,7 @@ fn wrap_text_with_ansi_impl( } let mut result = SmallVec::<[Vec; 4]>::new(); - let mut state = AnsiState::new(); + let mut state = WrapState::new(); let mut line_start = 0usize; for i in 0..=text.len() { diff --git a/crates/pi-natives/src/vectors.rs b/crates/pi-natives/src/vectors.rs new file mode 100644 index 000000000..5dbbc127c --- /dev/null +++ b/crates/pi-natives/src/vectors.rs @@ -0,0 +1,342 @@ +//! Batch numeric vector kernels for mnemopi recall paths. +//! +//! Every export processes an entire candidate batch per N-API crossing so the +//! crossing cost is amortized over the whole recall operation. Semantics +//! mirror the TypeScript reference implementations in +//! `packages/mnemopi/src/core` exactly — same accumulation order, same +//! non-finite handling, same tie-breaking — so float scores are +//! bit-identical to the TS versions and integer results are exactly equal. + +use napi::{ + Error, Result, Status, + bindgen_prelude::{Float32Array, Float64Array, Uint32Array}, +}; +use napi_derive::napi; + +fn invalid(message: &str) -> Result { + Err(Error::new(Status::InvalidArg, message)) +} + +#[inline] +const fn finite_or_zero(value: f64) -> f64 { + if value.is_finite() { value } else { 0.0 } +} + +/// Cosine similarity with the exact semantics of mnemopi's TS +/// `cosineSimilarity`: iterate `max(len_a, len_b)` elements, treat missing +/// and non-finite entries as `0`, return `0` when either norm is zero. +/// +/// Splitting the shared prefix from the tails preserves bit-exactness: tail +/// terms of the shorter side only ever add `±0.0` to `dot` and `+0.0` to its +/// own norm, in the same index order as the TS loop. +#[inline] +#[allow( + clippy::suboptimal_flops, + reason = "mul_add rounds differently; bit-exact with the TS loops is the contract" +)] +fn cosine_one(a: &[f64], b: &[f64]) -> f64 { + if a.is_empty() && b.is_empty() { + return 0.0; + } + let shared = a.len().min(b.len()); + let mut dot = 0.0f64; + let mut norm_a = 0.0f64; + let mut norm_b = 0.0f64; + for i in 0..shared { + let av = finite_or_zero(a[i]); + let bv = finite_or_zero(b[i]); + dot += av * bv; + norm_a += av * av; + norm_b += bv * bv; + } + for &raw in &a[shared..] { + let av = finite_or_zero(raw); + norm_a += av * av; + } + for &raw in &b[shared..] { + let bv = finite_or_zero(raw); + norm_b += bv * bv; + } + if norm_a == 0.0 || norm_b == 0.0 { + return 0.0; + } + dot / (norm_a.sqrt() * norm_b.sqrt()) +} + +/// All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. +/// +/// `vectors` is `count` vectors flattened row-major at `dim` `f64` elements +/// per row (zero-padded, which matches the TS `?? 0` missing-element +/// semantics), so the similarity is bit-identical to the TS pairwise loop in +/// `clusterBySimilarity`. Returns pairs flattened as `[i0, j0, i1, j1, ...]` +/// in the same `(i, j)` visit order as the TS nested loop. +#[napi] +pub fn cosine_similarity_pairs( + vectors: Float64Array, + count: u32, + dim: u32, + threshold: f64, +) -> Result { + let count = count as usize; + let dim = dim as usize; + let data: &[f64] = &vectors; + if data.len() != count * dim { + return invalid("vectors length must equal count * dim"); + } + let widened: &[f64] = data; + let mut pairs: Vec = Vec::new(); + for i in 0..count { + let left = &widened[i * dim..(i + 1) * dim]; + for j in (i + 1)..count { + let right = &widened[j * dim..(j + 1) * dim]; + if cosine_one(left, right) >= threshold { + pairs.push(i as u32); + pairs.push(j as u32); + } + } + } + Ok(Uint32Array::new(pairs)) +} + +/// Top-k rows of a normalized vector matrix ranked by dot product with a +/// normalized query. +#[napi(object)] +pub struct VectorTopK { + /// Row indices of the selected hits, best score first. + pub indices: Uint32Array, + /// Scores aligned with `indices`. + pub scores: Float64Array, +} + +/// Score every row of a normalized `f32` matrix against `query` and return +/// the top `limit` rows. +/// +/// Mirrors the TS `searchExactVectorIndex` loop bit-exactly: the query is +/// normalized by the L2 norm of its *full* length, each row score sums +/// `matrix[row][col] * (query[col] / norm)` over +/// `min(query.len, dimensions)` columns in column order. Ranking matches the +/// TS stable sort: score descending, lower row index first on exact ties +/// (`-0.0` and `+0.0` compare equal). Callers are expected to enforce the TS +/// guards first (finite query with a positive norm, non-empty matrix). +#[napi] +#[allow( + clippy::suboptimal_flops, + reason = "mul_add rounds differently; bit-exact with the TS loops is the contract" +)] +pub fn vector_index_top_k( + matrix: Float32Array, + dimensions: u32, + query: Float64Array, + limit: u32, +) -> Result { + let dims = dimensions as usize; + let data: &[f32] = &matrix; + if dims == 0 || !data.len().is_multiple_of(dims) { + return invalid("matrix length must be a positive multiple of dimensions"); + } + let count = data.len() / dims; + let q: &[f64] = &query; + let mut norm_sq = 0.0f64; + for &value in q { + norm_sq += value * value; + } + let norm = norm_sq.sqrt(); + // Hoisting the per-column division out of the row loop is bitwise + // identical to the TS per-row `query[col] / queryNorm`. + let query_dims = q.len().min(dims); + let normalized: Vec = q[..query_dims].iter().map(|&v| v / norm).collect(); + + let mut order: Vec<(f64, u32)> = Vec::with_capacity(count); + for row in 0..count { + let base = row * dims; + let mut score = 0.0f64; + for (col, &qv) in normalized.iter().enumerate() { + score += f64::from(data[base + col]) * qv; + } + order.push((score, row as u32)); + } + // JS comparator `(a, b) => b.score - a.score` under a stable sort: strict + // score ordering, otherwise (equal, including ±0.0) original row order. + order.sort_by(|a, b| { + let diff = b.0 - a.0; + if diff > 0.0 { + core::cmp::Ordering::Greater + } else if diff < 0.0 { + core::cmp::Ordering::Less + } else { + a.1.cmp(&b.1) + } + }); + let take = (limit as usize).min(order.len()); + order.truncate(take); + let indices: Vec = order.iter().map(|&(_, row)| row).collect(); + let scores: Vec = order.iter().map(|&(score, _)| score).collect(); + Ok(VectorTopK { indices: Uint32Array::new(indices), scores: Float64Array::new(scores) }) +} + +/// ECMA-262 `\s` (`WhiteSpace` ∪ `LineTerminator`), which differs from Rust's +/// `char::is_whitespace` (JS additionally includes U+FEFF). +#[inline] +const fn is_js_whitespace(c: char) -> bool { + matches!( + c, + '\u{0009}' + | '\u{000a}' + | '\u{000b}' + | '\u{000c}' + | '\u{000d}' + | '\u{0020}' + | '\u{00a0}' + | '\u{1680}' + | '\u{2000}' + ..='\u{200a}' + | '\u{2028}' + | '\u{2029}' + | '\u{202f}' + | '\u{205f}' + | '\u{3000}' + | '\u{feff}' + ) +} + +/// Lowercased word set per the TS `jaccardSimilarity` tokenizer: +/// `text.toLowerCase().split(/\s+/).filter(Boolean)` into a `Set`. +/// Returned sorted and deduplicated for merge-based intersection counting. +fn word_set(text: &str) -> Vec> { + let lower = text.to_lowercase(); + let mut words: Vec> = lower + .split(is_js_whitespace) + .filter(|w| !w.is_empty()) + .map(Box::from) + .collect(); + words.sort_unstable(); + words.dedup(); + words +} + +/// Jaccard similarity of two sorted, deduplicated word sets. Matches the TS +/// `jaccardSimilarity`: `0` when either set is empty, otherwise +/// `|A ∩ B| / (|A| + |B| - |A ∩ B|)` with exact integer counts. +fn jaccard_sorted(a: &[Box], b: &[Box]) -> f64 { + if a.is_empty() || b.is_empty() { + return 0.0; + } + let mut intersection = 0usize; + let (mut i, mut j) = (0usize, 0usize); + while i < a.len() && j < b.len() { + match a[i].cmp(&b[j]) { + core::cmp::Ordering::Less => i += 1, + core::cmp::Ordering::Greater => j += 1, + core::cmp::Ordering::Equal => { + intersection += 1; + i += 1; + j += 1; + }, + } + } + intersection as f64 / (a.len() + b.len() - intersection) as f64 +} + +/// MMR selection over pre-sorted candidates using Jaccard word similarity. +/// +/// `contents[i]` and `scores[i]` describe candidate `i`, already sorted by +/// relevance exactly as the TS `mmrRerank` sorts them (the JS stable sort +/// stays on the TS side so its tie and NaN semantics are preserved). +/// Replicates the TS selection loop exactly: candidate `0` is always taken +/// first; each round picks the remaining candidate maximizing +/// `lambda * score - (1 - lambda) * maxSimilarity(selected)` with strict +/// `>` comparisons, so ties keep the earliest remaining candidate — and a +/// round where every score is `NaN` picks the first remaining candidate, +/// matching the TS `bestIdx = 0` initialisation. Returns the selected +/// indices into the input order. +/// +/// Word tokenization matches `text.toLowerCase().split(/\s+/)` (ECMA `\s`, +/// Unicode default full case conversion). Known divergence: unpaired +/// surrogates arrive here as U+FFFD, while JS keeps the lone surrogate; both +/// tokenize to a single non-whitespace word so Jaccard counts still agree +/// unless a text mixes U+FFFD words with lone-surrogate words. +#[napi] +#[allow( + clippy::suboptimal_flops, + reason = "mul_add rounds differently; bit-exact with the TS loops is the contract" +)] +pub fn mmr_rerank_indices( + contents: Vec, + scores: Float64Array, + lambda_param: f64, + top_k: u32, +) -> Result { + if scores.len() != contents.len() { + return invalid("scores length must equal contents length"); + } + let limit = top_k as usize; + let count = contents.len(); + if limit == 0 || count == 0 { + return Ok(Uint32Array::new(Vec::new())); + } + let sets: Vec>> = contents.iter().map(|text| word_set(text)).collect(); + let mut selected: Vec = Vec::with_capacity(limit.min(count)); + selected.push(0); + let mut remaining: Vec = (1..count as u32).collect(); + + while !remaining.is_empty() && selected.len() < limit { + let mut best_idx = 0usize; + let mut best_score = f64::NEG_INFINITY; + for (idx, &candidate) in remaining.iter().enumerate() { + let mut max_similarity = 0.0f64; + for &picked in &selected { + let similarity = jaccard_sorted(&sets[candidate as usize], &sets[picked as usize]); + if similarity > max_similarity { + max_similarity = similarity; + } + } + let relevance = scores[candidate as usize]; + let mmr_score = lambda_param * relevance - (1.0 - lambda_param) * max_similarity; + if mmr_score > best_score { + best_score = mmr_score; + best_idx = idx; + } + } + selected.push(remaining.remove(best_idx)); + } + if selected.len() < limit { + selected.extend(remaining); + selected.truncate(limit); + } + Ok(Uint32Array::new(selected)) +} + +#[cfg(test)] +mod tests { + use super::{cosine_one, is_js_whitespace, jaccard_sorted, word_set}; + + #[test] + fn cosine_matches_reference_semantics() { + assert_eq!(cosine_one(&[], &[]), 0.0); + assert_eq!(cosine_one(&[1.0, 0.0], &[0.0, 0.0]), 0.0); + let same = cosine_one(&[1.0, 2.0, 3.0], &[1.0, 2.0, 3.0]); + assert!((same - 1.0).abs() < 1e-12); + // Non-finite entries are zeroed, mismatched lengths pad with zero. + let sim = cosine_one(&[f64::NAN, 1.0], &[0.5, 1.0, 2.0]); + let expect = 1.0 / (1.0f64.sqrt() * (0.25f64 + 1.0 + 4.0).sqrt()); + assert!((sim - expect).abs() < 1e-12); + } + + #[test] + fn word_set_matches_js_tokenizer() { + let set = word_set("Hello\u{00a0}WORLD hello\u{feff}world"); + assert_eq!(set, vec![Box::from("hello"), Box::from("world")]); + assert!(word_set("").is_empty()); + assert!(word_set(" \t\n").is_empty()); + assert!(!is_js_whitespace('\u{200b}')); // ZWSP is not JS \s + } + + #[test] + fn jaccard_matches_reference() { + let a = word_set("the quick brown fox"); + let b = word_set("the lazy brown dog"); + let sim = jaccard_sorted(&a, &b); + assert!((sim - 2.0 / 6.0).abs() < 1e-12); + assert_eq!(jaccard_sorted(&a, &word_set("")), 0.0); + } +} diff --git a/crates/pi-shell/src/minimizer/filters/mod.rs b/crates/pi-shell/src/minimizer/filters/mod.rs index d68963a89..3a0e21489 100644 --- a/crates/pi-shell/src/minimizer/filters/mod.rs +++ b/crates/pi-shell/src/minimizer/filters/mod.rs @@ -376,6 +376,7 @@ fn uv_wrapper_tool<'a>(ctx: &'a MinimizerCtx<'_>) -> Option<&'a str> { /// is already a single flag token and needs no entry here. const WRAPPER_VALUE_OPTIONS: &[&str] = &[ // uv run + "--extra", "--with", "--with-requirements", "--with-editable", @@ -633,7 +634,7 @@ mod tests { #[test] fn uv_run_pytest_routes_to_python_filter() { let config = MinimizerConfig::default(); - let context = ctx("uv", Some("run"), "uv run pytest", &config); + let context = ctx("uv", Some("run"), "uv run --extra turso pytest", &config); let input = "============================= test session starts \ ==============================\ncollected 2 items\n\na.py .\nb.py \ F\n\n=================================== FAILURES \ diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index 60e3967c2..f9bc30fb6 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -1141,11 +1141,16 @@ async fn run_shell_command_once( } } }); + // Let pipeline consumers flush output after cancellation kills their + // producers. The outer run cancellation remains bounded, and this delayed + // fallback still releases readers whose writers never close. + const CANCEL_READER_GRACE: Duration = Duration::from_millis(500); let cancel_bridge = tokio::spawn({ let cancel_token = cancel_token.clone(); let reader_cancel = reader_cancel.clone(); async move { cancel_token.cancelled().await; + time::sleep(CANCEL_READER_GRACE).await; reader_cancel.cancel(); } }); @@ -2205,6 +2210,50 @@ mod tests { let _ = std::fs::remove_dir_all(&tmp); } + /// Regression test for issue #5819: `mkdir -p ~/proj/{a,b}` must create both + /// `a` and `b` under `$HOME/proj`. Brace expansion runs before tilde + /// expansion and previously left every element after the first with a + /// literal `~`, so `b` was created as `./~/proj/b` in the shell cwd instead. + #[tokio::test(flavor = "multi_thread")] + async fn uutils_mkdir_expands_tilde_for_every_brace_element() { + let base = std::env::temp_dir().join(format!("pi-mkdir-brace-{}", std::process::id())); + let home = base.join("home"); + let cwd = base.join("cwd"); + let _ = std::fs::remove_dir_all(&base); + std::fs::create_dir_all(&home).expect("home dir"); + std::fs::create_dir_all(&cwd).expect("cwd dir"); + let cwd_str = cwd.to_str().expect("utf8 cwd path"); + + let mut env = HashMap::new(); + env.insert("HOME".to_string(), home.to_string_lossy().to_string()); + let config = + ShellConfig { session_env: Some(env), snapshot_path: None, minimizer: None }; + let mut session = create_session(&config).await.expect("create_session"); + session.shell.set_working_dir(cwd_str).expect("set cwd"); + + let mut params = session.shell.default_exec_params(); + params.set_fd(OpenFiles::STDIN_FD, null_file().expect("null")); + params.set_fd(OpenFiles::STDOUT_FD, null_file().expect("null")); + params.set_fd(OpenFiles::STDERR_FD, null_file().expect("null")); + + let source_info = SourceInfo::from("pi-natives:test"); + let exec = session + .shell + .run_string("mkdir -p ~/proj/{a,b}", &source_info, ¶ms) + .await + .expect("run_string"); + assert!(matches!(exec.exit_code, ExecutionExitCode::Success), "exit {}", exit_code(&exec)); + + // Both elements' tildes expanded: dirs land under $HOME/proj. + assert!(home.join("proj/a").is_dir(), "~/proj/a not created under HOME"); + assert!(home.join("proj/b").is_dir(), "~/proj/b not created under HOME"); + // The buggy path created a literal `~` tree in the shell cwd. + assert!(!cwd.join("~").exists(), "literal ~ tree leaked into cwd"); + assert!(!cwd.join("a").exists(), "unexpanded element leaked into cwd"); + + let _ = std::fs::remove_dir_all(&base); + } + /// `mkdir --help` and an invalid flag must be handled in-process: rendered /// to the command streams and returned as an exit code. The upstream /// `uumain` parser calls `std::process::exit`, which would terminate the diff --git a/crates/vendor/brush-core/src/expansion.rs b/crates/vendor/brush-core/src/expansion.rs index 6129fe79f..39e4d0d31 100644 --- a/crates/vendor/brush-core/src/expansion.rs +++ b/crates/vendor/brush-core/src/expansion.rs @@ -626,30 +626,44 @@ impl<'a, SE: extensions::ShellExtensions> WordExpander<'a, SE> { return Ok(Expansion::from(ExpansionPiece::Splittable(word.to_owned()))); } - // Apply brace expansion first, before anything else (not applicable to heredoc - // bodies). - let brace_expanded = self.brace_expand_if_needed(word)?; + // Apply brace expansion first, before anything else (not applicable to + // heredoc bodies). Each resulting element is an independent word: bash + // runs tilde/parameter/command/arithmetic expansion on EVERY element, so + // a tilde that begins any element (e.g. `~/{a,b}` -> `~/a` and `~/b`) + // must expand — not only the first. Parsing the space-joined result as a + // single word left every element after the first with a literal leading + // `~` (issue #5819). + let brace_expanded_words = self.brace_expand_words(word)?; if tracing::enabled!(target: trace_categories::EXPANSION, tracing::Level::DEBUG) - && brace_expanded != word + && !(brace_expanded_words.len() == 1 && brace_expanded_words[0] == word) { - tracing::debug!(target: trace_categories::EXPANSION, " => brace expanded to '{brace_expanded}'"); + tracing::debug!(target: trace_categories::EXPANSION, " => brace expanded to {brace_expanded_words:?}"); } - // Expand: tildes, parameters, command substitutions, arithmetic. - let pieces = if self.heredoc_mode { - // Heredoc mode only affects top-level parsing (literal quotes); recursive - // expansion of parameter words (e.g., ${var:-"default"}) uses normal semantics. - self.heredoc_mode = false; - - brush_parser::word::parse_heredoc(brace_expanded.as_ref(), &self.parser_options)? - } else { - brush_parser::word::parse(brace_expanded.as_ref(), &self.parser_options)? - }; - + // Expand each brace element separately (tildes, parameters, command + // substitutions, arithmetic), separating elements with a splittable + // space so downstream field splitting yields one field per element. let mut expansions = vec![]; - for piece in pieces { - let piece_expansion = self.expand_word_piece(piece.piece).await?; - expansions.push(piece_expansion); + for (index, element) in brace_expanded_words.iter().enumerate() { + if index > 0 { + expansions.push(Expansion::from(ExpansionPiece::Splittable(String::from(" ")))); + } + + let pieces = if self.heredoc_mode { + // Heredoc mode only affects top-level parsing (literal quotes); + // recursive expansion of parameter words (e.g., ${var:-"default"}) + // uses normal semantics. + self.heredoc_mode = false; + + brush_parser::word::parse_heredoc(element.as_ref(), &self.parser_options)? + } else { + brush_parser::word::parse(element.as_ref(), &self.parser_options)? + }; + + for piece in pieces { + let piece_expansion = self.expand_word_piece(piece.piece).await?; + expansions.push(piece_expansion); + } } let coalesced = coalesce_expansions(expansions); @@ -695,7 +709,13 @@ impl<'a, SE: extensions::ShellExtensions> WordExpander<'a, SE> { } } - fn brace_expand_if_needed(&self, word: &'a str) -> Result, error::Error> { + /// Perform brace expansion on `word`, returning each expanded element as a + /// separate word. When brace expansion doesn't apply (disabled, no braces, + /// or a parse failure), the original word is returned as the sole element. + /// + /// Empty brace elements (e.g. from `{,b}`) are returned as a quoted empty + /// string (`""`) so they survive as empty fields, matching bash. + fn brace_expand_words(&self, word: &'a str) -> Result>, error::Error> { // We perform a non-authoritative check to see if the string *may* contain // braces to expand. There may be false positives, but must be no false // negatives. @@ -703,28 +723,40 @@ impl<'a, SE: extensions::ShellExtensions> WordExpander<'a, SE> { || !self.shell.options().perform_brace_expansion || !may_contain_braces_to_expand(word) { - return Ok(word.into()); + return Ok(vec![word.into()]); } let parse_result = brush_parser::word::parse_brace_expansions(word, &self.parser_options); if parse_result.is_err() { tracing::error!("failed to parse for brace expansion: {parse_result:?}"); - return Ok(word.into()); + return Ok(vec![word.into()]); } let brace_expansion_pieces = parse_result?; let Some(brace_expansion_pieces) = brace_expansion_pieces else { - return Ok(word.into()); + return Ok(vec![word.into()]); }; tracing::debug!(target: trace_categories::EXPANSION, "Brace expansion pieces: {brace_expansion_pieces:?}"); - let result = braceexpansion::generate_and_combine_brace_expansions(brace_expansion_pieces) + let words = braceexpansion::generate_and_combine_brace_expansions(brace_expansion_pieces) .into_iter() - .map(|s| if s.is_empty() { "\"\"".into() } else { s }) - .join(" "); + .map(|s| if s.is_empty() { Cow::Borrowed("\"\"") } else { Cow::Owned(s) }) + .collect(); - Ok(result.into()) + Ok(words) + } + + /// Convenience wrapper over [`Self::brace_expand_words`] that joins the + /// expanded elements back into a single space-separated string. + #[cfg(test)] + fn brace_expand_if_needed(&self, word: &'a str) -> Result, error::Error> { + let mut words = self.brace_expand_words(word)?; + if words.len() == 1 { + Ok(words.pop().unwrap()) + } else { + Ok(Cow::Owned(words.join(" "))) + } } /// Apply tilde-expansion, parameter expansion, command substitution, and @@ -2082,6 +2114,32 @@ mod tests { Ok(()) } + /// Regression test for issue #5819: a tilde that begins each element of a + /// brace expansion must expand independently. Brace expansion joins its + /// elements before the tilde/parameter/... pass, so `~/{a,b}` must yield + /// `/a` and `/b` — not `/a` followed by a literal `~/b`. + #[tokio::test] + async fn test_tilde_expands_for_every_brace_element() -> Result<()> { + let mut shell = crate::shell::Shell::builder().build().await?; + shell + .env_mut() + .set_global("HOME", ShellVariable::new(ShellValue::String("/home/user".to_string())))?; + let params = shell.default_exec_params(); + + assert_eq!( + full_expand_and_split_word(&mut shell, ¶ms, "~/project/{a,b}").await?, + vec!["/home/user/project/a", "/home/user/project/b"], + ); + // A bare `~/{a,b}` (tilde immediately followed by the brace) must also + // expand on both elements. + assert_eq!( + full_expand_and_split_word(&mut shell, ¶ms, "~/{a,b}").await?, + vec!["/home/user/a", "/home/user/b"], + ); + + Ok(()) + } + #[tokio::test] async fn test_field_splitting() -> Result<()> { let mut shell = crate::shell::Shell::builder().build().await?; diff --git a/crates/vendor/uu-rm/src/rm.rs b/crates/vendor/uu-rm/src/rm.rs index 0420973ac..8c2b37697 100644 --- a/crates/vendor/uu-rm/src/rm.rs +++ b/crates/vendor/uu-rm/src/rm.rs @@ -545,6 +545,12 @@ fn create_progress_bar(files: &[&OsStr], recursive: bool) -> Option fn count_files(paths: &[&OsStr], recursive: bool) -> u64 { let mut total = 0; for p in paths { + // Empty operands are rejected by `remove()` before deletion; skip them + // here too so the progress pre-count doesn't resolve "" to the cwd and + // walk the entire working directory into the total. + if p.is_empty() { + continue; + } let path = Path::new(p); if let Ok(md) = fs::symlink_metadata(pi_uutils_ctx::resolve(path)) { if md.is_dir() && !is_symlink_dir(&md) { @@ -595,6 +601,19 @@ pub fn remove(files: &[&OsStr], options: &Options) -> bool { for filename in files { let file = Path::new(filename); + // An empty operand can never name a real file. Guard it before + // `pi_uutils_ctx::resolve`, which joins "" onto the shell's working + // directory — without this, `rm -rf ""` resolves to the cwd and + // recursively deletes it. GNU rm reports ENOENT for an empty operand + // (and `rm -f` stays silent), so mirror that here. + if filename.is_empty() { + if !options.force { + show_error!("{}", RmError::CannotRemoveNoSuchFile(filename.to_os_string())); + had_err = true; + } + continue; + } + // Check if the path (potentially with trailing slash) resolves to root // This needs to happen before symlink_metadata to catch cases like "rootlink/" // where rootlink is a symlink to root. @@ -1106,4 +1125,63 @@ mod tests { assert_eq!(Path::new("/"), clean_trailing_slashes(path)); } + + /// Regression: `rm -rf ""` must never delete the shell working directory. + /// + /// An empty operand used to reach `pi_uutils_ctx::resolve`, which joins "" + /// onto the cwd and yields the cwd itself, so `-rf` recursively removed it + /// (issue #6287). The builtin now rejects the empty operand before + /// resolution, leaving the cwd and its contents intact. + #[test] + fn empty_operand_does_not_delete_cwd() { + use std::{ + collections::HashMap, + ffi::OsString, + sync::{ + Arc, + atomic::{AtomicBool, AtomicU32, Ordering}, + }, + time::{SystemTime, UNIX_EPOCH}, + }; + + use pi_uutils_ctx::{ScopeIo, scope}; + + // Unique disposable working directory with a sentinel file inside it. + static COUNTER: AtomicU32 = AtomicU32::new(0); + let nanos = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_nanos(); + let seq = COUNTER.fetch_add(1, Ordering::Relaxed); + let cwd = std::env::temp_dir().join(format!( + "omp-rm-empty-{}-{}-{}", + std::process::id(), + nanos, + seq + )); + std::fs::create_dir_all(&cwd).unwrap(); + let sentinel = cwd.join("sentinel"); + std::fs::write(&sentinel, b"keep me").unwrap(); + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(std::io::sink()), + stderr: Box::new(std::io::sink()), + cwd: cwd.clone(), + env: HashMap::new(), + cancel: Arc::new(AtomicBool::new(false)), + }; + + let args = vec![OsString::from("rm"), OsString::from("-rf"), OsString::new()]; + let code = scope(io, || crate::run(args)); + + // `rm -f` swallows the empty-operand error, matching GNU rm's exit 0. + assert_eq!(code, 0, "rm -rf \"\" should exit 0 under --force"); + assert!(cwd.is_dir(), "working directory must survive rm -rf \"\""); + assert!(sentinel.is_file(), "sentinel must survive rm -rf \"\""); + + std::fs::remove_dir_all(&cwd).ok(); + } } diff --git a/crates/vendor/uu-tail/src/tail.rs b/crates/vendor/uu-tail/src/tail.rs index 83ae6e229..4b5c596c4 100644 --- a/crates/vendor/uu-tail/src/tail.rs +++ b/crates/vendor/uu-tail/src/tail.rs @@ -633,16 +633,12 @@ fn unbounded_tail(reader: &mut BufReader, settings: &Settings) -> UR }, _ => {}, } - #[cfg(not(target_os = "windows"))] + // pi-uutils: upstream emulates Unix SIGPIPE on Windows by calling + // `std::process::exit(13)` on a broken-pipe flush. That would kill the + // long-lived host shell process. An in-process builtin must never + // `process::exit`; let the broken pipe surface as a normal `io::Error` and + // propagate to the caller, matching every other pi-uutils builtin. writer.flush()?; - - // SIGPIPE is not available on Windows. - #[cfg(target_os = "windows")] - writer.flush().inspect_err(|err| { - if err.kind() == ErrorKind::BrokenPipe { - std::process::exit(13); - } - })?; Ok(()) } @@ -855,4 +851,47 @@ mod tests { assert_ne!(code, 0, "broken pipe must surface as a non-zero exit, not a panic"); } + + #[test] + fn unbounded_tail_broken_pipe_does_not_abort() { + use std::{ + collections::HashMap, + ffi::OsString, + io::{self, Cursor, ErrorKind, Write}, + sync::{Arc, atomic::AtomicBool}, + }; + + // Same broken-pipe consumer as above, but here stdin is a plain reader + // so `tail_stdin` always takes the streaming `unbounded_tail` path — the + // one the reported repro (`seq ... | tail -n 3 | head -n 0`) exercises, + // and where the Windows SIGPIPE emulation used to `std::process::exit`. + struct BrokenPipeWriter; + impl Write for BrokenPipeWriter { + fn write(&mut self, _buf: &[u8]) -> io::Result { + Err(io::Error::new(ErrorKind::BrokenPipe, "Broken pipe")) + } + + fn flush(&mut self) -> io::Result<()> { + Err(io::Error::new(ErrorKind::BrokenPipe, "Broken pipe")) + } + } + + let input = b"1\n2\n3\n4\n5\n".to_vec(); + let io = pi_uutils_ctx::ScopeIo { + stdin: Box::new(Cursor::new(input)), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(BrokenPipeWriter), + stderr: Box::new(io::sink()), + cwd: std::env::temp_dir(), + env: HashMap::new(), + cancel: Arc::new(AtomicBool::new(false)), + }; + + let code = pi_uutils_ctx::scope(io, || { + crate::run(vec![OsString::from("tail"), OsString::from("-n"), OsString::from("3")]) + }); + + assert_ne!(code, 0, "broken pipe must surface as a non-zero exit, not process::exit"); + } } diff --git a/docs/advisor-watchdog.md b/docs/advisor-watchdog.md index 65870bf11..8fdd17542 100644 --- a/docs/advisor-watchdog.md +++ b/docs/advisor-watchdog.md @@ -38,13 +38,24 @@ advisor: The advisor role uses normal model-role resolution, including provider-prefixed ids, canonical ids, and optional thinking suffixes. +### Headless runs + +Use `--advisor` to enable the advisor for one print-mode process without +persisting `advisor.enabled`: + +```sh +omp -p --advisor "Review this task." +``` + +While a primary prompt is running, advisor concerns and blockers continue to steer that live turn. After the final prompt settles, print mode preserves late advisor notes without starting hidden primary turns, then waits up to ten minutes for final reviews before disposing the session. Error exits use a 30-second drain budget so failed automation can terminate. If either deadline expires, OMP logs the reviews that disposal will abandon; completed reviews retain their transcript and token/cost usage. + Slash commands: | Command | Effect | |---|---| -| `/advisor` | Toggle the persisted `advisor.enabled` setting. | -| `/advisor on` | Enable the setting and start the runtime when an advisor model is assigned. | -| `/advisor off` | Disable the setting and stop the runtime. | +| `/advisor` | Toggle the advisor for this session (session-scoped override; does not change the persisted `advisor.enabled` setting). | +| `/advisor on` | Enable the advisor for this session and start the runtime when an advisor model is assigned. Session-scoped; not persisted to config. | +| `/advisor off` | Disable the advisor for this session and stop the runtime. Session-scoped; not persisted to config. | | `/advisor status` | Show active model, context usage, token usage, and cost. | | `/advisor dump` | Copy the advisor's compact transcript to the clipboard. | | `/advisor dump raw` | Copy the advisor's full dump (system prompt, tools, thinking, and calls) to the clipboard. | @@ -89,8 +100,8 @@ The `advise` tool accepts one note and an optional severity: | Severity | Delivery | Intended use | |---|---|---| | omitted / `nit` | Non-interrupting aside, batched into the primary transcript at the next step boundary. | Cleanup, simplification, low-risk edge cases. | -| `concern` | Interrupting steering message. | Material risk, likely wrong direction, missing constraint, hallucinated API. | -| `blocker` | Interrupting steering message. | Continuing would clearly waste work or produce broken output. | +| `concern` | Interrupting steering message when the delivery constraints below permit it. A late terminal-answer `concern` is preserved as a visible card instead. | Material risk, likely wrong direction, missing constraint, hallucinated API. | +| `blocker` | Interrupting steering message when the delivery constraints below permit it. Unlike a `concern`, a terminal answer alone does not prevent it from triggering a turn. | Continuing would clearly waste work or produce broken output. | Interrupting advice is sent through the steering channel and can abort in-flight tools at the next steering boundary. Each note (interrupting or batched) is rendered into the primary transcript as an `` element — severity rides a `severity` attribute, and a `guidance` attribute carries the "weigh, don't blindly obey" framing (the primary agent's system prompt never mentions advisories, so the tag is its only cue). Note bodies are XML-escaped so advice containing `<`, `>`, or `&` can't break the wrapper: @@ -100,7 +111,21 @@ note text ``` -When you deliberately interrupt the agent (Esc, or a cancel from collab, ACP, RPC, the SDK, or an extension), the advisor stops auto-resuming it. An interrupting `concern`/`blocker` raised while the run is stopped is recorded as a visible advisor card instead of restarting the turn, and a concern already in flight when you interrupt is preserved the same way rather than driving a surprise resume. The advice re-enters context the next time you resume — a new message, the `.`/`c` continue shortcut, or a steer/follow-up. A normal yield is unaffected: the advisor can still steer and resume a run the agent ended on its own. +When you deliberately interrupt the agent (Esc, or a cancel from collab, ACP, RPC, the SDK, or an extension), the advisor stops auto-resuming it. An interrupting `concern`/`blocker` raised while the run is stopped is recorded as a visible advisor card instead of restarting the turn, and a concern already in flight when you interrupt is preserved the same way rather than driving a surprise resume. The advice re-enters context the next time you resume — a new message, the `.`/`c` continue shortcut, or a steer/follow-up. + +A normal yield the agent drove itself is treated differently from a deliberate interrupt, but it is not a blanket "always steers and resumes". The loop state and completed turn first determine the normal delivery path: + +- **While the loop is still streaming** (the raise arrived before the yield, or during a resume you already drove), the note normally steers into the live turn. +- **Once the loop has yielded and gone idle**, delivery keys on how the turn ended: + - If the primary's tail is a **terminal text answer with no queued work**, a late `concern` is preserved as a visible card rather than waking the agent to restate a completed turn (#4840) — it re-enters context on the next resume (a new message, `.`/`c`, or a steer/follow-up), exactly like the interrupt case. A `blocker` is the exception: it normally steers a triggered turn, because it means the agent handed off broken or unexercised work that must be acknowledged before the turn is considered done (#5628). + - Otherwise (the agent yielded mid-work, no terminal answer), an idle `concern`/`blocker` normally triggers a fresh turn so the advice is acted on immediately. + +Two session/client constraints can still preserve a note whose normal delivery path is steering: + +- **Plan mode:** every would-be advisor steer is preserved as a visible card, even while the primary loop is streaming, because only user-driven turns converge on ask/resolve. +- **ACP with deferred agent-initiated turns:** when `deferAgentInitiatedTurns` is enabled and the bridge has not allowed agent-initiated turns, an idle would-be steer is preserved because the client cannot represent the triggered turn as busy. Advice raised while the primary loop is already streaming can still steer into that live turn. + +So the advisor can steer and resume a run the agent ended on its own **while it is running or yielded mid-work and the current mode/client permits steering**. When steering is blocked instead, the note is either preserved as a card (the terminal-answer, plan-mode, and deferred-ACP cases above) or downgraded to a non-interrupting aside (the `advisor.immuneTurns` cooldown below); either way it waits for the next step boundary or resume rather than waking the agent. `advisor.immuneTurns` limits interruption frequency. After the advisor successfully delivers a `concern` or `blocker` through the steering channel, later concerns/blockers are routed as non-interrupting asides until the configured number of primary turns has completed. The default is `3`. `nit` notes are unchanged, and advice raised while user-interrupt auto-resume suppression is active is still preserved instead of restarting a stopped run. diff --git a/docs/environment-variables.md b/docs/environment-variables.md index adf3d310c..d146a97f1 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -351,12 +351,13 @@ Extra conditional behavior: ## 6) Storage and config root paths -These are consumed via `@oh-my-pi/pi-utils/dirs` and affect where coding-agent stores data. +These affect where coding-agent stores data and which process-local settings overlays it loads. | Variable | Default / behavior | | --------------------- | ----------------------------------------------------------------------------- | | `PI_CONFIG_DIR` | Config root dirname under home (default `.omp`) | | `PI_CODING_AGENT_DIR` | Full override for agent directory (default `~//agent`) | +| `PI_CONFIG_FILES` | Platform path-list of settings overlays (`:` on Unix, `;` on Windows); loaded in order before explicit `--config` overlays | | `PWD` | Used when matching canonical current working directory in path helpers | --- diff --git a/docs/extensions.md b/docs/extensions.md index 786f66284..0da8c250e 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -160,6 +160,30 @@ Handlers and tool `execute` receive `ctx` with: - `shutdown()` - `getSystemPrompt()` - `memory` (optional structured memory runtime — status/search/save across the configured backend) +- `setInterval(fn, ms, ...args)` / `setTimeout(fn, ms, ...args)` / `clearTimer(timer)` — managed timers (see below) + +### Background work (`ctx.setInterval` / `ctx.setTimeout`) + +Extensions run **in-process with no isolation**. A raw `setInterval`/`setTimeout`/detached-promise callback that throws runs outside the handler-dispatch try/catch, surfaces as a process-level `uncaughtException`, and the global postmortem handler treats it as fatal — **the whole session is torn down**, not just the offending extension. + +Use `ctx.setInterval` / `ctx.setTimeout` for any periodic or deferred background work. They mirror the platform signatures but: + +- run the callback with the same isolation as handler dispatch — a synchronous throw or a rejected promise is logged and reported through the extension error channel, and the session keeps running; +- return a handle you can pass to `ctx.clearTimer(handle)`; +- are `unref`'d (never keep the process alive on their own) and are cleared automatically on `session_shutdown`. + +```ts +pi.on("session_start", async (_event, ctx) => { + const timer = ctx.setInterval(() => { + // A throw here is contained — it will not crash the session. + ctx.ui.notify("tick", "info"); + }, 60_000); + // Optional: clear it yourself; otherwise it is cleared on shutdown. + pi.on("session_shutdown", () => ctx.clearTimer(timer)); +}); +``` + +If you use raw `setInterval`/`setTimeout` or detached promises instead, you own the isolation: wrap the callback body in your own `try/catch` (an unhandled throw will take down the session) and clear the timer on `session_shutdown`. ### Model selection (`ctx.models`) diff --git a/docs/magic-keywords.md b/docs/magic-keywords.md new file mode 100644 index 000000000..19750d4da --- /dev/null +++ b/docs/magic-keywords.md @@ -0,0 +1,46 @@ +# Magic keywords + +Magic keywords are standalone words in a user prompt that add a hidden instruction for that turn. They are enabled by default and glow in the editor when `omp` recognizes them. + +## Keywords + +| Keyword | Effect | +|---|---| +| `ultrathink` | Asks the agent to reason carefully through a multi-step task. When automatic thinking is active, it also selects the highest reasoning effort supported by the current model for that turn. | +| `orchestrate` | Switches the agent to the multi-agent orchestration contract: scope the full task, delegate substantial independent work in parallel, verify each phase, and continue until the request is complete. | +| `workflowz` | Asks the agent to build and run a deterministic multi-subagent workflow with the `task` tool. It is intended for broad research, reviews, migrations, or other work that benefits from parallel coverage. The keyword only adds its instruction when `task` is available in the active tool set. | + +Use the keyword anywhere in the prose of the prompt: + +```text +ultrathink about the failure modes before changing this API + +orchestrate the migration described in docs/plan.md + +workflowz an adversarial review of the authentication changes +``` + +## Matching rules + +Matching is deliberate so source code and paths do not accidentally change agent behavior: + +- Use the exact lowercase spelling. `Ultrathink`, `Orchestrate`, and `Workflowz` do not trigger. +- The keyword must be standalone. Sentence punctuation may touch it, but identifiers, inflections, paths, and file extensions do not match. For example, `orchestrate,` matches; `orchestrated` and `orchestrate.ts` do not. +- Fenced code blocks, inline code spans, and XML/HTML sections are ignored. +- The instruction applies to the user turn containing the keyword. The highlighted word remains part of the visible prompt; the added instruction is hidden. + +## Configuration + +Open `/settings` and use **Interaction → Magic Keywords**, or change the settings from a shell: + +```bash +# Disable every magic keyword +omp config set magicKeywords.enabled false + +# Disable one keyword while leaving the others enabled +omp config set magicKeywords.ultrathink false +omp config set magicKeywords.orchestrate false +omp config set magicKeywords.workflow false +``` + +All four settings default to `true`. Run `omp config list` to inspect every available setting and its current value. See [Settings](./settings.md) for configuration scopes, precedence, and project-local overrides. diff --git a/docs/models.md b/docs/models.md index 61335e182..e77cfb840 100644 --- a/docs/models.md +++ b/docs/models.md @@ -300,7 +300,9 @@ When `litellm` is active (for example through `LITELLM_API_KEY` or stored auth), - base URL: explicit provider `baseUrl` / `models.yml` config, otherwise `LITELLM_BASE_URL`, otherwise `http://localhost:4000/v1` - auth mode: `LITELLM_API_KEY` or stored LiteLLM auth when the proxy requires a key -Runtime discovery probes LiteLLM management metadata first: `GET /model_group/info`, then `GET /v2/model/info`, then falls back to the OpenAI-compatible `GET /models` list. Rich metadata maps `max_input_tokens`, `max_output_tokens`, `supports_vision`, and `supports_reasoning`; bare fallback ids are enriched against bundled reference metadata when available. +Runtime discovery probes LiteLLM management metadata in order: `GET /model_group/info`, `GET /v2/model/info`, `GET /model/info`, and `GET /v1/model/info`. The configured key must be authorized to read at least one of these routes; on deployments that restrict management endpoints, grant the route through LiteLLM's `allowed_routes` access controls or use a master/admin key for discovery. + +If every metadata route is unavailable, discovery falls back to the OpenAI-compatible `GET /models` list. A forbidden or failed metadata request is logged once with its endpoint and status; `404` is treated as an absent route. Rich metadata maps per-model context and capability fields, while bare fallback ids are enriched against bundled reference metadata when available. Models absent from the bundled catalog can therefore have unknown context and pricing after fallback. ### Explicit provider discovery diff --git a/docs/providers.md b/docs/providers.md index 4481157d7..f41fda858 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -31,7 +31,7 @@ When a provider needs an API key, `omp` resolves it in this order (first match w 1. **Runtime override** — a key supplied for the current process, e.g. CLI `--api-key`. Never persisted. 2. **`models.yml` config key** — an `apiKey` pinned on a custom provider, registered as a config-sourced bearer. This deliberately beats stored OAuth, so a key supplied for a custom `baseUrl`/gateway is honored instead of forwarding an upstream OAuth token the proxy would reject. 3. **Stored API key** — an API-key credential saved in the auth store. -4. **Stored OAuth credential** — refreshed when needed; multiple accounts are ranked/rotated automatically. For Anthropic, each organization counts as its own account: one email holding both a Team seat and a personal plan can log in once per subscription (pick the workspace on the browser consent page) and rotation treats them as two accounts. +4. **Stored OAuth credential** — refreshed when needed; multiple accounts are ranked/rotated automatically. For Anthropic and ChatGPT (Codex), each organization/workspace counts as its own account: one email holding both a Team/Enterprise seat and a personal plan can log in once per subscription (pick the workspace on the browser consent page) and rotation treats them as two accounts. 5. **Provider environment variable** — including values loaded from `.env` files (see [the env-var table](#environment-variables-and-env-files)). 6. **`models.yml` fallback resolver** — keys for custom providers not otherwise registered. diff --git a/docs/rpc.md b/docs/rpc.md index 973b80e1b..f1f27db30 100644 --- a/docs/rpc.md +++ b/docs/rpc.md @@ -25,15 +25,48 @@ Behavior notes: - RPC mode disables automatic session title generation by default to avoid an extra model call. - RPC mode resets workflow-altering `todo.*`, `task.*`, `memory.backend`/`memories.enabled`, `advisor.*`, `async.*`, and `bash.autoBackground.*` settings to their built-in defaults instead of inheriting user overrides. - The process reads stdin as JSONL (`readJsonl(Bun.stdin.stream())`). -- At startup it writes `{ "type": "ready" }` before processing commands. +- At startup it writes a `ready` frame before processing commands. The frame advertises supported protocol versions and transport limits. - When stdin closes, pending host-tool calls and host-URI requests are rejected and the process exits with code `0`. - Responses/events are written as one JSON object per line. ## Transport and Framing -Each frame is a single JSON object followed by `\n`. +Protocol v1 frames are a single JSON object followed by `\n`. Every physical JSONL frame is limited to 1 MiB. -There is no envelope beyond the object shape itself. +The initial ready frame uses protocol v1 and advertises the opt-in lossless transport: + +```json +{ + "type": "ready", + "protocolVersion": 1, + "supportedProtocolVersions": [1, 2], + "maxFrameBytes": 1048576, + "maxReassembledFrameBytes": 67108864 +} +``` + +Clients that support protocol v2 SHOULD immediately send: + +```json +{ "id": "protocol-1", "type": "negotiate_protocol", "protocolVersion": 2 } +``` + +After the success response, oversized stdout objects are emitted losslessly as an uninterrupted sequence of `rpc_chunk` frames. Each chunk carries a base64 segment of the original UTF-8 JSON object: + +```json +{ + "type": "rpc_chunk", + "chunkId": "rpc-1", + "index": 0, + "count": 7, + "byteLength": 1600042, + "data": "eyJ0eXBlIjoicmVzcG9uc2UiLC4uLn0=" +} +``` + +Clients MUST validate `chunkId`, `index`, `count`, and `byteLength`, reject interleaved or interrupted sequences, enforce the advertised reassembly limit, concatenate decoded bytes in index order, decode them as strict UTF-8, and parse the result as one JSON object. The exported TypeScript `RpcFrameDecoder` implements this validation. The bundled TypeScript and Python `RpcClient` implementations negotiate v2 automatically when the ready frame advertises it. + +Legacy clients may ignore the added ready fields and remain on v1. V1 retains its bounded fallback behavior for oversized output. Frames above the v2 reassembly ceiling still fail explicitly; large history APIs should use pagination rather than depending on arbitrarily large logical frames. ### Outbound frame categories (stdout) @@ -84,6 +117,10 @@ Important edge behavior from runtime: - `{ id?, type: "abort_and_prompt", message: string, images?: ImageContent[] }` - `{ id?, type: "new_session", parentSession?: string }` +### Protocol + +- `{ id?, type: "negotiate_protocol", protocolVersion: 2 }` + ### State - `{ id?, type: "get_state" }` @@ -148,6 +185,11 @@ correlate it via `id`. Ordering across concurrent commands is not guaranteed ### Messages - `{ id?, type: "get_messages" }` +- `{ id?, type: "get_messages_page", cursor?: string, limit?: number }` + +`get_messages_page` returns a stable chronological page with `messages`, `totalMessages`, and an opaque `nextCursor` when more messages remain. Cursors are bound to the session ID, durable leaf, and message count. The server rejects stale cursors if the session changes between requests, and refuses to start a paging walk while the session is streaming or compacting. Failed page requests carry a machine-readable `code` on the error response — `session_busy` (session is streaming or compacting) or `stale_cursor` (the snapshot behind the cursor changed, e.g. a background bash appended a message between pages) — so clients can react without matching error-message text. Pages contain at most 256 messages and normally stay below the v1 physical-frame ceiling. A v1 caller can page ordinary histories, but an individual message whose response exceeds that ceiling produces an overflow error; retrieving it losslessly requires negotiated v2 framing. + +The bundled TypeScript `RpcClient.getMessages()` and Python `RpcClient.get_messages()` drain this paged endpoint automatically after negotiating v2. They retain the legacy monolithic command when connected to a v1 server, and on either `session_busy` or `stale_cursor` they discard partial pages and fall back to the legacy best-effort snapshot. Direct `getMessagesPage()` and `get_messages_page()` calls remain strict so incremental hosts never mix snapshots silently. ### Login diff --git a/docs/settings.md b/docs/settings.md index 682279460..eecd5e942 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -8,6 +8,7 @@ Settings are stored as plain YAML mappings. Every key, its type, default, and en - For custom model definitions in `models.yml`, see [Models](./models.md). - For instruction files discovered into the agent context (`AGENTS.md`, `.omp/`, etc.), see [Context files](./context-files.md). - For the full catalog of environment variables, see [Environment variables](./environment-variables.md). +- For prompt words that activate specialized per-turn behavior, see [Magic keywords](./magic-keywords.md). ## Where settings live @@ -121,6 +122,7 @@ Environment variables are **not** a single settings layer. Each is read by the f | `OMP_AUTH_BROKER_URL` | `auth.broker.url` | Env value takes precedence over config. | | `OMP_AUTH_BROKER_TOKEN` | `auth.broker.token` | Env value takes precedence over config. | | `PI_CODING_AGENT_DIR` | (relocates agent dir) | Moves `config.yml`, `agent.db`, and the whole agent base. | +| `PI_CONFIG_FILES` | CLI config overlays | Platform path-list (`:` on Unix, `;` on Windows); files load in order before `--config` overlays. | Provider API keys are resolved separately (stored auth, OAuth, `models.yml`, environment, and `.env` files); see [Providers](./providers.md) and the full [Environment variables](./environment-variables.md) reference. @@ -216,6 +218,10 @@ omp --config ./local/ci-settings.yml "check this failure" omp --config ./base.yml --config ./experiment.yml "try this model" ``` +`--config` is accepted by the default launch command, `acp`, and `models`. + +Wrappers may instead set `PI_CONFIG_FILES` to a platform-delimited path list (`:` on Unix, `;` on Windows). Environment overlays load in listed order before explicit `--config` overlays. + Overlay paths are resolved relative to the process working directory (and `~` is expanded). Each overlay must parse as a YAML mapping; a missing file, invalid YAML, or a top-level array/scalar is a hard error — it does **not** silently fall back to lower-precedence settings. ## Path-scoped arrays diff --git a/docs/skills/authoring-extensions.md b/docs/skills/authoring-extensions.md index b904eddec..9522e5262 100644 --- a/docs/skills/authoring-extensions.md +++ b/docs/skills/authoring-extensions.md @@ -246,6 +246,7 @@ The derived name is the filename stem (or directory name for `index.ts`-style en - **Do not call runtime actions during load.** Methods like `pi.sendMessage()` throw `ExtensionRuntimeNotInitializedError` if called synchronously during module evaluation (before a session is active). Register handlers/tools/commands during load; perform runtime actions only from event handlers, tools, or commands. - **`tool_call` errors are fail-closed.** If a `tool_call` handler throws, the tool is blocked. +- **Self-scheduled callbacks run in-process with no isolation.** A raw `setInterval`/`setTimeout`/detached-promise callback that throws escapes the handler-dispatch try/catch and crashes the whole session (`uncaughtException`). Use `ctx.setInterval` / `ctx.setTimeout` for background work — they contain callback throws and auto-clear on `session_shutdown`. With raw timers you must add your own `try/catch` and cleanup. - **Command names must not clash with built-ins.** Conflicts are skipped with a diagnostic log. - **Reserved shortcuts are ignored** (`ctrl+c`, `ctrl+d`, `ctrl+z`, `ctrl+k`, `ctrl+p`, `ctrl+l`, `ctrl+o`, `ctrl+t`, `ctrl+g`, `ctrl+q`, `alt+m`, `shift+tab`, `shift+ctrl+p`, `alt+enter`, `escape`, `enter`). diff --git a/docs/task-agent-discovery.md b/docs/task-agent-discovery.md index 4ce802152..7962f8b67 100644 --- a/docs/task-agent-discovery.md +++ b/docs/task-agent-discovery.md @@ -122,16 +122,26 @@ In spawn execution (`TaskTool.#executeSync` → `#runSpawn`): `TaskTool.create()` builds the tool description from discovery results at initialization time. `#executeSync` rediscovers agents, so the runtime set can differ from what was listed in the earlier tool description if agent files changed mid-session. The async entry path still uses the initialization-time list to decide whether an agent is marked `blocking` before scheduling. -## Structured-output guardrails and schema precedence +## Model and structured-output precedence -Runtime output schema precedence in `TaskTool.#runSpawn`: +Runtime model precedence is resolved by `resolveEffectiveSubagentPolicy()`: -1. agent frontmatter `output` -2. parent session `outputSchema` +1. the task item's explicit `model` selector or fallback chain +2. `task.agentModelOverrides[agentName]` +3. agent frontmatter `model` +4. the parent session model fallback -(`effectiveOutputSchema = effectiveAgent.output ?? this.session.outputSchema` — the task call itself never carries a schema; ad-hoc structured workflows go through the eval bridge's `agent(prompt, schema)`.) +Reasoning suffixes such as `:high` are preserved. The task tool rejects blank, empty-array, and comma-only per-call selectors before policy resolution so they cannot bypass a configured agent override. -The model-facing prompt (`src/prompts/tools/task.md`) no longer carries the old structured-output mismatch warning; it tags read-only agents and warns against offloading reasoning to `explore`/`sonic` instead. +Runtime output schema precedence is: + +1. the task item's explicit `outputSchema` +2. agent frontmatter `output` +3. parent session `outputSchema` + +The task item's optional `schemaMode` overrides the parent session mode; the default is `permissive`. + +The model-facing prompt (`src/prompts/tools/task.md`) no longer carries the old structured-output mismatch warning; it tags read-only agents and warns against offloading reasoning to `scout`/`sonic` instead. ## Command discovery interaction diff --git a/docs/tools/debug.md b/docs/tools/debug.md index ebb874f2c..e0670b49f 100644 --- a/docs/tools/debug.md +++ b/docs/tools/debug.md @@ -122,19 +122,19 @@ Side-channel artifacts outside the model tool result: 1. Tool registration is conditional: `DebugTool.createIf()` in `packages/coding-agent/src/tools/debug.ts` returns `null` unless `session.settings.get("debug.enabled")` is true. `packages/coding-agent/src/tools/index.ts` wires the factory and rechecks the same setting in tool filtering. 2. `DebugTool.execute()` clamps `params.timeout` through `clampTimeout("debug", params.timeout)` and composes the caller `AbortSignal` with `AbortSignal.timeout(...)`. 3. `launch` and `attach` resolve cwd/program paths, select an adapter in `packages/coding-agent/src/dap/config.ts`, then delegate to `dapSessionManager.launch()` / `.attach()`. -4. `DapSessionManager.launch()` / `.attach()` enforce the single-session rule with `#ensureLaunchSlot()`, spawn the adapter through `DapClient.spawn()`, register listeners, send `initialize`, cache capabilities, start listening for an initial stop event before sending `launch`/`attach`, then complete the `initialized` → `configurationDone` handshake in `#completeConfigurationHandshake()`. -5. `DapClient.spawn()` starts the adapter detached with `NON_INTERACTIVE_ENV`. Most adapters use stdio; socket-mode adapters (`dlv`) use `#spawnSocketUnix()` on Linux or `#spawnSocketClientAddr()` on macOS/other. +4. `DapSessionManager.launch()` / `.attach()` enforce one root session, spawn the adapter through `DapClient.spawn()`, register listeners, send `initialize`, cache capabilities, subscribe for tree-wide stop events, send `launch`/`attach`, then complete the `initialized` → `configurationDone` handshake. +5. `DapClient.spawn()` starts adapters detached with `NON_INTERACTIVE_ENV`. Most adapters use stdio; socket-mode adapters (`dlv`) use an adapter-specific Unix/TCP transport, while TCP server adapters start with `${port}` substituted in their args. Child sessions reuse the root TCP server through `DapClient.connect()`. 6. `#registerSession()` in `packages/coding-agent/src/dap/session.ts` installs reverse-request handlers: - `runInTerminal`: spawns the requested debuggee command detached via `ptree.spawn()` and returns `{ processId }` - - `startDebugging`: logs the child-session request and returns `{}`; it does not create nested sessions - - events: `output`, `initialized`, `stopped`, `continued`, `exited`, `terminated` update cached session state -7. Operational actions (`set_breakpoint`, `evaluate`, `threads`, `read_memory`, `custom_request`, and similar) call `dapSessionManager` methods. Most flow through `#sendRequestWithConfig()`, which first sends `configurationDone` when required, then sends the DAP request, then updates `lastUsedAt`. -8. Breakpoint actions maintain local cached breakpoint sets in `DapSessionManager` and remap adapter responses back onto those cached records. -9. `continue` and the three step actions clear cached stop state, subscribe for `stopped`/`terminated`/`exited` before sending the DAP request, then `#awaitStopOutcome()` either returns the new stopped location or reports that the program is still running after timeout. + - `startDebugging`: connects a child DAP client to the root TCP server, forwards the requested `launch`/`attach` configuration, binds root breakpoints before `configurationDone`, and recursively installs the same handlers + - events: `output`, `initialized`, `stopped`, `continued`, `exited`, and `terminated` update cached session state; stopped children become the active target +7. Operational actions (`set_breakpoint`, `evaluate`, `threads`, `read_memory`, `custom_request`, and similar) call `dapSessionManager` methods. Most flow through `#sendRequestWithConfig()`, which first sends `configurationDone` when required, then sends the DAP request and refreshes the active session plus its ancestors. +8. Breakpoint actions synchronize desired breakpoint sets across the live root/child tree. New children receive those sets before their `configurationDone` request. +9. `continue` and the three step actions clear cached stop state, subscribe for a stop/termination event anywhere in the session tree before sending the DAP request, then `#awaitStopOutcome()` returns the active child’s stopped location or reports that the target remains running after timeout. 10. `pause` sends DAP `pause`, waits for a stopped event if needed, and reuses cached stop state if the program was already stopped. -11. `stack_trace`, `scopes`, `variables`, and `evaluate` default to the current stopped thread/frame when the caller omits ids and cached state is available. -12. `output` reads the in-memory output ring from `DapSessionManager.getOutput()`. `terminate` sends `terminate` when supported, always attempts `disconnect`, marks the session terminated, and disposes the client. -13. `sessions` reads the manager’s current map and formats all summaries. Although the manager stores a map, only one active session can exist because new launch/attach calls are blocked until the active one is terminated or cleaned up. +11. `stack_trace`, `scopes`, `variables`, and `evaluate` default to the current stopped child/thread/frame when the caller omits ids and cached state is available. +12. `output` reads the in-memory output ring from the active `DapSession`. `terminate` walks from the root through every child, sends best-effort `terminate`/`disconnect`, and disposes the complete tree even when an adapter times out. +13. `sessions` reads the manager’s current map and formats root and child summaries. Only one root tree can exist; recursive adapter-requested children are tracked with `parentSessionId` / `childSessionIds`. 14. The interactive selector in `packages/coding-agent/src/debug/index.ts` builds a `SelectList` of fixed values and dispatches each to a handler: - `performance`: `startCpuProfile()`, wait for Enter/Escape, stop profiling, read a 30-second work profile with `getWorkProfile(30)`, then bundle via `createReportBundle()` - `work`: read `getWorkProfile(30)`, write a temp SVG, open it externally @@ -320,12 +320,13 @@ Example `.omp/dap.json`: - `collectSystemInfo()` is best-effort for CPU probing; failure there falls back to `Unknown CPU`. ## Notes -- `packages/coding-agent/src/prompts/tools/debug.md` tells the model only one active session is supported; that is not advisory, it is enforced in code. -- `configurationDone` is sent automatically both during launch/attach handshake and lazily before later requests if the adapter required it and the initial handshake did not complete. -- `startDebugging` reverse requests are acknowledged but not implemented; child debug sessions are not spawned. -- `output` exposes the merged `output` event stream only; the tool does not distinguish stdout, stderr, and console categories. -- Session summaries expose `needsConfigurationDone`; this is derived from adapter capabilities and whether `configurationDone` has been sent. -- Source breakpoint file paths are normalized with `path.resolve()` before caching and sending to the adapter. +- `packages/coding-agent/src/prompts/tools/debug.md` tells the model only one active root session is supported. Adapter-requested child sessions belong to that root tree. +- The default JavaScript/TypeScript adapter runs vscode-js-debug’s `dapDebugServer.js` over TCP. Install it with Mason or set `JS_DEBUG_DAP_SERVER` to a release-tarball server path. +- `configurationDone` is sent automatically during root and child launch/attach handshakes and lazily before later requests if the initial handshake did not complete. +- `startDebugging` reverse requests create recursive child sessions on the same TCP server; a stopped child becomes the target for thread-level actions. +- `output` exposes the active session’s merged `output` event stream only; the tool does not distinguish stdout, stderr, and console categories. +- Session summaries expose `needsConfigurationDone`, `parentSessionId`, and `childSessionIds`. +- Source breakpoint file paths are normalized with `path.resolve()` before caching and synchronizing across the tree. - `evaluate` defaults to `repl`, so the tool can forward raw debugger commands when the adapter supports them. - `disassemble` resolves its target from `memory_reference` first, then the current stopped session's `instructionPointerReference`; it throws if neither is present. - `RawSseDebugBuffer.recordEvent()` increments `totalEvents` before bounded retention. A snapshot can therefore show fewer retained records than total observed events. diff --git a/docs/tools/lsp.md b/docs/tools/lsp.md index ca4be1f86..f86f9e058 100644 --- a/docs/tools/lsp.md +++ b/docs/tools/lsp.md @@ -31,7 +31,7 @@ | `query` | string | No | Workspace symbol query, code-action selector/filter, or LSP method name for `action=request`. | | `new_name` | string | No | Required for `rename` and `rename_file`. | | `apply` | boolean | No | For `rename`/`rename_file`, apply unless explicitly `false`. For `code_actions`, list unless explicitly `true`. | -| `timeout` | number | No | Seconds, clamped by `clampTimeout("lsp", ...)` to `5..60`, default `20`. | +| `timeout` | number | No | Seconds, clamped by `clampTimeout("lsp", ...)` to `5..300`, default `20`. | | `payload` | string | No | JSON string for `action=request`; overrides auto-built params. | ## Outputs @@ -268,7 +268,7 @@ Same as `definition`, but sends `textDocument/implementation` and reports `imple - Background message readers persist for each live client until process exit/shutdown. ## Limits & Caps -- Tool timeout clamp: default `20`, min `5`, max `60` seconds — `TOOL_TIMEOUTS.lsp` in `packages/coding-agent/src/tools/tool-timeouts.ts`. +- Tool timeout clamp: default `20`, min `5`, max `300` seconds — `TOOL_TIMEOUTS.lsp` in `packages/coding-agent/src/tools/tool-timeouts.ts`. - LSP request default timeout inside `sendRequest()`: `30_000ms` — `DEFAULT_REQUEST_TIMEOUT_MS` in `packages/coding-agent/src/lsp/client.ts`. - Warmup initialize timeout default: `5_000ms` — `WARMUP_TIMEOUT_MS` in `packages/coding-agent/src/lsp/client.ts`. - Project-load wait fallback: `15_000ms` — `PROJECT_LOAD_TIMEOUT_MS` in `packages/coding-agent/src/lsp/client.ts`. diff --git a/docs/tools/task.md b/docs/tools/task.md index 1a5a98be9..3d7883489 100644 --- a/docs/tools/task.md +++ b/docs/tools/task.md @@ -26,9 +26,9 @@ ## Inputs -The wire schema is shape-swapped by `task.batch` (default on). One unit of work is the task item `{ name?, agent?, task, isolated? }` (`isolated` only when `task.isolation.mode` is not `none`): +The wire schema is shape-swapped by `task.batch` (default on). One unit of work is the task item `{ name?, agent?, task, model?, outputSchema?, schemaMode?, isolated? }` (`isolated` only when `task.isolation.mode` is not `none`): -- **Batch shape** (`task.batch` on): `{ context, tasks: item[] }` — one subagent per item, all run under the same fan-out rules; there is no top-level agent field. `context` is **required** shared background rendered into every spawned subagent's system prompt (`CONTEXT` section); `agent` and `isolated` are per item, so one call may mix agent types. +- **Batch shape** (`task.batch` on): `{ context, tasks: item[] }` — one subagent per item, all run under the same fan-out rules; there is no top-level agent field. `context` is **required** shared background rendered into every spawned subagent's system prompt (`CONTEXT` section); `agent`, `model`, `outputSchema`, `schemaMode`, and `isolated` are per item, so one call may mix agent types, models, and output contracts. - **Flat shape** (`task.batch` off): `{ ...item }` — exactly one spawn per call. Shared background goes into a `local://` file (e.g. `local://ctx.md`) that each spawn's `task` references; subagents share the parent's `local://` root. | Field | Type | Required | Description | @@ -38,13 +38,16 @@ The wire schema is shape-swapped by `task.batch` (default on). One unit of work | `name` | `string` | No | Stable agent name — becomes the registry/IRC id. Defaults to a generated AdjectiveNoun name. Uniquified per session by `AgentOutputManager`. Item field in batch shape, top-level in flat shape. | | `agent` | `string` | No | Agent type to run this item (e.g. `scout`). Defaults to the spawn policy's default agent (usually `task`); items in one batch call may use different agent types. Item field in batch shape, top-level in flat shape. | | `task` | `string` | Yes | The work — complete, self-contained instructions. Empty-after-trim is rejected. Item field in batch shape, top-level in flat shape. | +| `model` | `string \| string[]` | No | Explicit non-empty model selector or non-empty fallback chain for this spawn. Optional `:reasoning` suffixes are preserved. Takes precedence over `task.agentModelOverrides` and agent frontmatter. Item field in batch shape, top-level in flat shape. | +| `outputSchema` | JSON Schema object | No | Invocation-specific structured-output contract. Takes precedence over agent frontmatter `output` and the inherited parent session schema. Item field in batch shape, top-level in flat shape. | +| `schemaMode` | `"permissive" \| "strict"` | No | Validation mode for the effective output schema. Overrides the parent session mode; defaults to `permissive`. Item field in batch shape, top-level in flat shape. | | `isolated` | `boolean` | No | Run in an isolated workspace and return patches. Exists only when `task.isolation.mode` is not `none`; per item in batch shape, top-level in flat shape. Isolated agents are torn down at completion — not revivable. | There is no wire label field: the one-line UI label shown in the TUI/registry is generated automatically from the `task` text by the tiny/title model (fire-and-forget), so callers never provide it. Runtime stays permissive: the flat form is accepted even while `task.batch` is on (internal callers such as the commit flow's `analyze_files`, and stale transcripts). The model only ever sees one shape. -There is no per-call `schema` parameter. Structured output comes from the agent definition's `output` frontmatter, the inherited parent session schema, or — for ad-hoc workflows — the eval bridge's `agent(prompt, schema)`. +There is no legacy per-call `schema` parameter. Use `outputSchema` and optional `schemaMode`; when absent, structured output falls back to the agent definition's `output` frontmatter and then the inherited parent session schema. ## Outputs @@ -68,13 +71,13 @@ Settled response (`async.enabled=false`, no job manager, every item's agent `blo Artifacts and side channels: - Every subagent with an artifacts dir writes `.md`; `agent://` resolves to that file. -- If the output file is JSON, `agent:///` and `agent://?q=` perform JSON extraction. +- A subagent's own children are dot-qualified (`.`); `agent:///` reads that nested output. When the path names no nested output and the file is JSON, `agent:///` and `agent://?q=` perform JSON extraction. - Each subagent gets `.jsonl` session history when the parent persists artifacts; `history://` renders it as a concise transcript (works for live and parked agents). - Isolated patch mode writes `.patch` before merge. ## Flow 1. `TaskTool.create(...)` discovers agents once per cwd through a process-level memo (`discoverAgentsForCreate`) to render the dynamic prompt description. -2. `execute(...)` repairs raw params (`repairTaskParams`), then validates: `schema` is always rejected; `tasks`/`context` are rejected unless `task.batch` is on; batch calls need a non-empty `tasks` (a `task` per item, unique provided names), a non-empty shared `context`, and no top-level `task` alongside `tasks`; flat calls need `task`. The call is then normalized into its spawn list (`resolveSpawnItems`). +2. `execute(...)` repairs raw params (`repairTaskParams`), then validates: `schema` is always rejected; model selectors that normalize to no patterns are rejected; `tasks`/`context` are rejected unless `task.batch` is on; batch calls need a non-empty `tasks` (a `task` per item, unique provided names), a non-empty shared `context`, and no top-level `task` alongside `tasks`; flat calls need `task`. The call is then normalized into its spawn list (`resolveSpawnItems`). 3. Per-item execution split: items whose agent type declares `blocking: true` run inline; the rest become background jobs. The whole call runs sync when `async.enabled=false`, the session has no `AsyncJobManager` (orphaned host), or every item is blocking; inline spawns run through `#executeSync(...)` under the session-scoped semaphore. 4. Background execution (any non-blocking item with `async.enabled=true` and an `AsyncJobManager`): - agent ids are allocated up front via `AgentOutputManager.allocate(...)` — each item's `name`, or a generated AdjectiveNoun name — one per spawn; @@ -84,7 +87,7 @@ Artifacts and side channels: - a mixed call registers the async jobs first, then runs its blocking items inline and returns once they settle — the text combines the inline summaries with the spawned-job listing, and the block keeps rendering the still-running background rows beside the inline results. 5. `#executeSync(...)` runs the spawn path (`#runSpawn`), which rediscovers agents from disk, so runtime resolution can differ from the create-time description. 6. It resolves each spawn's requested `agent` type, rejects unknown or settings-disabled agents, and enforces parent spawn policy plus `PI_BLOCKED_AGENT` self-recursion prevention. -7. Output schema priority: agent frontmatter `output` → inherited parent session schema (the call itself never carries one). +7. Model priority: per-call `model` → `task.agentModelOverrides` → agent frontmatter → configured task role/session fallback. Output schema priority: per-call `outputSchema` → agent frontmatter `output` → inherited parent session schema. 8. Plan mode swaps in an `effectiveAgent` with a read-only tool subset and plan-mode prompt; `runSubprocess(...)` receives the effective agent. 9. If `isolated`, it requires a git repo (`getRepoRoot(...)` / `captureBaseline(...)`), maps `task.isolation.mode` to a backend-kind hint (`parseIsolationMode`), and materializes the workspace via the natives PAL (`ensureIsolation` → `isoResolve`/`isoStart`), walking the candidate list when a backend is unavailable. 10. Artifacts dir comes from the parent session file when available, otherwise a temp dir. When the session is executing an approved plan, the plan reference is handed to the subagent. @@ -103,7 +106,7 @@ Artifacts and side channels: - Background job — `async.enabled=true`; non-blocking spawns go through `AsyncJobManager`. - Sync inline — `async.enabled=false`, no job manager, or the item's agent declares `blocking: true` (per item: a mixed call runs both modes). - Batch mode (`task.batch`, default on) - - on — `{ context, tasks[] }`: one independent spawn per item, required `context` shared across the call's spawns, `agent`/`isolated` per item. Lifecycle, revival, and concurrency semantics match N parallel single calls. + - on — `{ context, tasks[] }`: one independent spawn per item, required `context` shared across the call's spawns, with `agent`, `model`, `outputSchema`, `schemaMode`, and `isolated` per item. Lifecycle, revival, and concurrency semantics match N parallel single calls. - off — single spawn per call; `tasks`/`context` are rejected and removed from the schema. - Isolation mode (`task.isolation.mode`): `none`, `auto`, `apfs`, `btrfs`, `zfs`, `reflink`, `overlayfs`, `projfs`, `block-clone`, `rcopy` (legacy `worktree`, `fuse-overlay`, `fuse-projfs` accepted for back-compat); the PAL resolves the actual backend with fallback. - Isolation merge strategy: patch mode (capture/apply root patches) or branch mode (commit to `omp/task/`, cherry-pick into parent). diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md index 25dfb4ad0..53b535a5e 100644 --- a/docs/tui-core-renderer.md +++ b/docs/tui-core-renderer.md @@ -135,6 +135,10 @@ of history: - `getNativeScrollbackLiveRegionStart()` — first row that may still mutate (everything below it, including root chrome rendered after it, stays in the window). +- `isNativeScrollbackLiveRegionPinned()` — optional policy for replacing + dashboards: rows at/after the live boundary stay viewport-local instead of + entering history as frozen snapshots. When the boundary advances or + disappears, newly final rows commit in order. - `getNativeScrollbackCommitSafeEnd()` — optional **byte-stable** deeper boundary (B): the append-only prefix of the live region (a streaming assistant message's settled rows), asserted never to re-layout, so it stays under the audit. diff --git a/package.json b/package.json index 06675e521..e4365a3cc 100644 --- a/package.json +++ b/package.json @@ -26,24 +26,29 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.2", - "@oh-my-pi/hashline": "17.0.0", - "@oh-my-pi/omp-stats": "17.0.0", - "@oh-my-pi/pi-agent-core": "17.0.0", - "@oh-my-pi/pi-ai": "17.0.0", - "@oh-my-pi/pi-catalog": "17.0.0", - "@oh-my-pi/pi-coding-agent": "17.0.0", - "@oh-my-pi/pi-mnemopi": "17.0.0", - "@oh-my-pi/pi-natives": "17.0.0", - "@oh-my-pi/pi-tui": "17.0.0", - "@oh-my-pi/pi-utils": "17.0.0", - "@oh-my-pi/pi-wire": "17.0.0", - "@oh-my-pi/snapcompact": "17.0.0", + "@oh-my-pi/hashline": "17.0.8", + "@oh-my-pi/omp-stats": "17.0.8", + "@oh-my-pi/pi-agent-core": "17.0.8", + "@oh-my-pi/pi-ai": "17.0.8", + "@oh-my-pi/pi-catalog": "17.0.8", + "@oh-my-pi/pi-coding-agent": "17.0.8", + "@oh-my-pi/pi-mnemopi": "17.0.8", + "@oh-my-pi/pi-natives": "17.0.8", + "@oh-my-pi/pi-tui": "17.0.8", + "@oh-my-pi/pi-utils": "17.0.8", + "@oh-my-pi/pi-wire": "17.0.8", + "@oh-my-pi/snapcompact": "17.0.8", "@opentelemetry/api": "^1.9.1", - "@opentelemetry/context-async-hooks": "^2.7.1", + "@opentelemetry/api-logs": "^0.220.0", + "@opentelemetry/context-async-hooks": "^2.9.0", + "@opentelemetry/exporter-logs-otlp-proto": "^0.220.0", + "@opentelemetry/exporter-metrics-otlp-proto": "^0.220.0", "@opentelemetry/exporter-trace-otlp-proto": "^0.220.0", - "@opentelemetry/resources": "^2.7.1", - "@opentelemetry/sdk-trace-base": "^2.7.1", - "@opentelemetry/sdk-trace-node": "^2.7.1", + "@opentelemetry/resources": "^2.9.0", + "@opentelemetry/sdk-logs": "^0.220.0", + "@opentelemetry/sdk-metrics": "^2.9.0", + "@opentelemetry/sdk-trace-base": "^2.9.0", + "@opentelemetry/sdk-trace-node": "^2.9.0", "@puppeteer/browsers": "^3.0.6", "@tailwindcss/node": "^4.3.2", "@tailwindcss/vite": "^4.3.2", @@ -112,7 +117,7 @@ "build:native": "bun --cwd=packages/natives run build", "test": "bun scripts/ci-test-ts.ts local", "test:ts": "bun scripts/ci-test-ts.ts local-ts", - "test:scripts": "bun test scripts/ci-build-native.test.ts scripts/ci-concurrency.test.ts scripts/ci-release-build-binaries.test.ts scripts/ci-release-notes.test.ts scripts/fix-dts-extensions.test.ts scripts/link-omp.test.ts", + "test:scripts": "bun test scripts/ci-build-native.test.ts scripts/ci-concurrency.test.ts scripts/ci-release-build-binaries.test.ts scripts/ci-release-notes.test.ts scripts/ci-release-publish.test.ts scripts/fix-dts-extensions.test.ts scripts/link-omp.test.ts scripts/musl-release.test.ts", "test:rs": "bun scripts/run-rs-task.ts test:rs", "check": "bun run --parallel check:ts check:rs", "check:ts": "bun run check:tools && bun run --workspaces --if-present check", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index e01b2e781..ac50d94dd 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,29 @@ ## [Unreleased] +## [17.0.8] - 2026-07-22 + +### Fixed + +- Improved resilience against transient stream JSON parse failures by recovering completed tool calls while safely preventing incomplete, unknown, refused, or sensitive calls from executing. + +## [17.0.5] - 2026-07-18 + +### Added + +- Added a per-message token estimation cache to optimize performance by reusing token counts for settled message history, with automatic cache invalidation on message mutation. + +### Changed + +- Improved tool execution control by making tool interruptibility resolvable per call, allowing side-effecting operations to complete while passive waits can yield to queued steering. + +## [17.0.2] - 2026-07-17 + +### Fixed + +- Improved error visibility in interactive clients by surfacing provider stream failures through the assistant message lifecycle, preventing silent loading spinners. +- Fixed an issue where Cursor provider contexts omitted host-supplied MCP tools from main and side-channel requests. + ## [17.0.0] - 2026-07-15 ### Breaking Changes diff --git a/packages/agent/package.json b/packages/agent/package.json index 1367d6eea..099dfe026 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "17.0.0", + "version": "17.0.8", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index a2400e44e..1dafe1dc9 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -183,6 +183,7 @@ type AssistantToolCallBlock = Extract block.type === "toolCall"); if (toolCalls.length === 0) return message; + const stopDetailType = message.stopDetails?.type; + const stopDetailCategory = message.stopDetails?.category; + if ( + stopDetailType === "refusal" || + stopDetailType === "sensitive" || + stopDetailCategory === "refusal" || + stopDetailCategory === "sensitive" + ) + return message; const availableToolNames = new Set(); for (const tool of availableTools) { availableToolNames.add(tool.name); if (tool.customWireName !== undefined) availableToolNames.add(tool.customWireName); } if (!toolCalls.every(toolCall => availableToolNames.has(toolCall.name))) return message; - if (!AIError.isStreamReadErrorText(`${message.errorMessage ?? ""}\n${message.stopDetails?.explanation ?? ""}`)) + if ( + !AIError.isStreamReadErrorText(`${message.errorMessage ?? ""}\n${message.stopDetails?.explanation ?? ""}`) && + !AIError.isTransientStreamParseError(message.errorMessage) && + !AIError.isTransientStreamParseError(message.stopDetails?.explanation) + ) return message; return { ...message, @@ -1821,11 +1837,25 @@ async function executeToolCalls( const tool = tools?.find(t => t.name === toolCall.name) ?? tools?.find(t => t.customWireName !== undefined && t.customWireName === toolCall.name); + const args = toolCall.arguments as Record; + const interruptibleMode = tool?.interruptible; + let interruptible = false; + if (typeof interruptibleMode === "function") { + try { + interruptible = interruptibleMode(args); + } catch { + // Resolver failures default to preserving the tool's outcome. + interruptible = false; + } + } else { + interruptible = interruptibleMode === true; + } return { toolCall, tool, - args: toolCall.arguments as Record, - signal: tool?.interruptible ? interruptibleSignal : nonInterruptibleSignal, + args, + interruptible, + signal: interruptible ? interruptibleSignal : nonInterruptibleSignal, started: false, result: undefined as AgentToolResult | undefined, isError: false, @@ -2207,16 +2237,16 @@ async function executeToolCalls( } } - // While an interruptible tool is in flight (e.g. a `job`/`irc` wait - // blocking on external work), queued steering or interrupting IRC would - // otherwise wait out the tool's own window. Poll only non-consuming queues - // and abort the shared tool signal so the boundary dequeue below injects - // the message promptly. Gated on immediate-interrupt mode + an - // interruptible tool; checkSteering is idempotent (no-op once triggered). + // While an interruptible tool call is in flight (e.g. a `hub` wait blocking + // on external work), queued steering or interrupting IRC would otherwise + // wait out the tool's own window. Poll only non-consuming queues and abort + // the shared tool signal so the boundary dequeue below injects the message + // promptly. Gated on immediate-interrupt mode + an interruptible call; + // checkSteering is idempotent (no-op once triggered). const watchSteeringWhileRunning = shouldInterruptImmediately && (hasSteeringMessages !== undefined || hasIrcInterrupts !== undefined) && - records.some(r => r.tool?.interruptible === true); + records.some(record => record.interruptible); const steeringWatchTimer = watchSteeringWhileRunning ? setInterval(() => void checkSteering(), STEERING_INTERRUPT_POLL_MS) : undefined; @@ -2268,6 +2298,23 @@ export interface SyntheticToolResultDetails { upstreamError?: string; } +/** + * Narrow an {@link AgentMessage} to a synthetic {@link ToolResultMessage} — + * a tool_result emitted for a tool call the assistant never invoked (see + * {@link SyntheticToolResultDetails}). Consumers use this to look past the + * placeholder pairing back to the assistant turn that produced it, e.g. + * `AgentSession.retry()` walking back over the synthetic results a + * stalled/aborted mid-tool-call turn leaves behind. + */ +export function isSyntheticToolResultMessage( + message: AgentMessage | undefined, +): message is ToolResultMessage { + return ( + message?.role === "toolResult" && + (message.details as SyntheticToolResultDetails | undefined)?.__synthetic === true + ); +} + function syntheticDetailsFor( reason: "aborted" | "error" | "skipped" | "length", errorMessage: string | undefined, @@ -2289,18 +2336,14 @@ function syntheticDetailsFor( } /** - * Create a tool result for a tool call that was emitted by the assistant but - * never invoked locally. Maintains the tool_use / tool_result pairing the - * provider API requires, and tags {@link SyntheticToolResultDetails} so - * consumers can distinguish this from a real local tool failure without - * string-matching the content (#4321). + * Create the persisted synthetic result for a tool call that was emitted by + * the assistant but never invoked locally. */ -function createAbortedToolResult( +export function createSyntheticToolResultMessage( toolCall: Extract, - stream: EventStream, reason: "aborted" | "error" | "skipped" | "length", errorMessage?: string, -): ToolResultMessage { +): ToolResultMessage { const message = reason === "aborted" ? "Tool execution was aborted" @@ -2310,9 +2353,31 @@ function createAbortedToolResult( ? "Tool call was not executed because the assistant ended its turn" : "Tool call was not executed because the provider stream ended with an error before the tool could run"; const details = syntheticDetailsFor(reason, errorMessage); - const result: AgentToolResult = { + return { + role: "toolResult", + toolCallId: toolCall.id, + toolName: toolCall.name, content: [{ type: "text", text: errorMessage ? `${message}: ${errorMessage}` : `${message}.` }], details, + isError: true, + timestamp: Date.now(), + }; +} + +/** + * Create and emit a tool result for a tool call that was emitted by the + * assistant but never invoked locally. + */ +function createAbortedToolResult( + toolCall: Extract, + stream: EventStream, + reason: "aborted" | "error" | "skipped" | "length", + errorMessage?: string, +): ToolResultMessage { + const toolResultMessage = createSyntheticToolResultMessage(toolCall, reason, errorMessage); + const result: AgentToolResult = { + content: toolResultMessage.content, + details: toolResultMessage.details, }; stream.push({ @@ -2329,17 +2394,6 @@ function createAbortedToolResult( result, isError: true, }); - - const toolResultMessage: ToolResultMessage = { - role: "toolResult", - toolCallId: toolCall.id, - toolName: toolCall.name, - content: result.content, - details, - isError: true, - timestamp: Date.now(), - }; - stream.push({ type: "message_start", message: toolResultMessage }); stream.push({ type: "message_end", message: toolResultMessage }); diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 280109497..090eb0de0 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -31,6 +31,7 @@ import { abortReasonText, agentLoop, agentLoopContinue, + createSyntheticToolResultMessage, normalizeMessagesForProvider, normalizeTools, resolveOwnedDialectFromEnv, @@ -64,12 +65,6 @@ function defaultConvertToLlm(messages: AgentMessage[]): Message[] { }); } -const ANTHROPIC_OUTPUT_BLOCKED_PREFIX = "Output blocked by conten"; - -function isAnthropicOutputBlockedError(message: string): boolean { - return message.includes(ANTHROPIC_OUTPUT_BLOCKED_PREFIX); -} - function refreshToolChoiceForActiveTools( toolChoice: ToolChoice | undefined, tools: AgentContext["tools"] = [], @@ -267,6 +262,8 @@ export interface AgentOptions { * Cursor exec handlers for local tool execution. */ cursorExecHandlers?: CursorExecHandlers; + /** Additional tools Cursor executes through its MCP request-context bridge, resolved before each provider call. */ + getCursorTools?: () => AgentTool[]; /** * Cursor tool result callback for exec tool responses. @@ -368,6 +365,7 @@ export class Agent { #maxRetryDelayMs?: number; #getToolContext?: (toolCall?: ToolCallContext) => AgentToolContext | undefined; #cursorExecHandlers?: CursorExecHandlers; + #getCursorTools?: () => AgentTool[]; #cursorOnToolResult?: CursorToolResultHandler; #cwd?: string; #cwdResolver?: () => string | undefined; @@ -450,6 +448,7 @@ export class Agent { this.#onSseEvent = opts.onSseEvent; this.#getToolContext = opts.getToolContext; this.#cursorExecHandlers = opts.cursorExecHandlers; + this.#getCursorTools = opts.getCursorTools; this.#cursorOnToolResult = opts.cursorOnToolResult; this.#cwd = opts.cwd; this.#cwdResolver = opts.cwdResolver; @@ -694,6 +693,22 @@ export class Agent { this.#appendOnlyContext = manager; } + #toolsForModel(model: Model): AgentTool[] { + if (model.api !== "cursor-agent" || !this.#getCursorTools) return this.#state.tools; + const cursorTools = this.#getCursorTools(); + if (cursorTools.length === 0) return this.#state.tools; + + const names = new Set(this.#state.tools.map(tool => tool.name)); + let merged: AgentTool[] | undefined; + for (const tool of cursorTools) { + if (names.has(tool.name)) continue; + merged ??= this.#state.tools.slice(); + merged.push(tool); + names.add(tool.name); + } + return merged ?? this.#state.tools; + } + /** * Assemble the provider Context for a side-channel (no-loop) request, mirroring * the main loop's prefix (system + normalized tools) so it shares the prompt @@ -718,7 +733,7 @@ export class Agent { const tools = ownedDialect ? [] : (normalizeTools( - this.#state.tools, + this.#toolsForModel(model), this.#intentTracing, preferredDialect(model.id), this.#pruneToolDescriptions, @@ -1152,7 +1167,7 @@ export class Agent { await Bun.sleep(0); } context.systemPrompt = this.#state.systemPrompt; - context.tools = this.#state.tools; + context.tools = this.#toolsForModel(this.#state.model ?? model); }, cursorExecHandlers: this.#cursorExecHandlers, cursorOnToolResult, @@ -1205,6 +1220,7 @@ export class Agent { }; let partial: AgentMessage | null = null; + const completedToolCallIds = new Set(); try { const stream = messages @@ -1222,6 +1238,9 @@ export class Agent { case "message_update": partial = event.message; this.#state.streamMessage = event.message; + if (event.assistantMessageEvent.type === "toolcall_end") { + completedToolCallIds.add(event.assistantMessageEvent.toolCall.id); + } break; case "message_end": @@ -1283,12 +1302,22 @@ export class Agent { : err instanceof Error ? err.message : String(err); - const shouldEmitVisibleOutputBlockedError = !stoppedForAbort && isAnthropicOutputBlockedError(errorMessage); + const shouldEmitVisibleError = !stoppedForAbort; const assistantPartial = partial?.role === "assistant" ? partial : undefined; const hadAssistantStart = assistantPartial !== undefined; + const bufferedCursorResults = this.#cursorToolResultBuffer.map(({ toolResult }) => toolResult); + const retainedToolCallIds = new Set(completedToolCallIds); + for (const { toolCallId } of bufferedCursorResults) retainedToolCallIds.add(toolCallId); const errorMsg: AssistantMessage = - shouldEmitVisibleOutputBlockedError && assistantPartial - ? { ...assistantPartial, stopReason: "error", errorMessage } + shouldEmitVisibleError && assistantPartial + ? { + ...assistantPartial, + content: assistantPartial.content.filter( + block => block.type !== "toolCall" || retainedToolCallIds.has(block.id), + ), + stopReason: "error", + errorMessage, + } : { role: "assistant", content: [{ type: "text", text: "" }], @@ -1308,7 +1337,7 @@ export class Agent { timestamp: Date.now(), }; - if (shouldEmitVisibleOutputBlockedError) { + if (shouldEmitVisibleError) { if (!hadAssistantStart) { this.#state.streamMessage = errorMsg; this.#emit({ type: "message_start", message: errorMsg }); @@ -1317,8 +1346,40 @@ export class Agent { this.appendMessage(errorMsg); this.#state.error = errorMessage; this.#emit({ type: "message_end", message: errorMsg }); - this.#emit({ type: "turn_end", message: errorMsg, toolResults: [] }); - this.#emit({ type: "agent_end", messages: [errorMsg] }); + const toolResults: ToolResultMessage[] = []; + this.#cursorToolResultBuffer = []; + const bufferedCursorToolCallIds = new Set(bufferedCursorResults.map(({ toolCallId }) => toolCallId)); + for (const toolResult of bufferedCursorResults) { + this.appendMessage(toolResult); + this.#emit({ type: "message_start", message: toolResult }); + this.#emit({ type: "message_end", message: toolResult }); + toolResults.push(toolResult); + } + for (const block of errorMsg.content) { + if (block.type !== "toolCall") continue; + if (bufferedCursorToolCallIds.has(block.id)) continue; + const toolResult = createSyntheticToolResultMessage(block, "error", errorMessage); + this.#emit({ + type: "tool_execution_start", + toolCallId: block.id, + toolName: block.name, + args: block.arguments, + intent: block.intent, + }); + this.#emit({ + type: "tool_execution_end", + toolCallId: block.id, + toolName: block.name, + result: { content: toolResult.content, details: toolResult.details }, + isError: true, + }); + this.appendMessage(toolResult); + this.#emit({ type: "message_start", message: toolResult }); + this.#emit({ type: "message_end", message: toolResult }); + toolResults.push(toolResult); + } + this.#emit({ type: "turn_end", message: errorMsg, toolResults }); + this.#emit({ type: "agent_end", messages: [errorMsg, ...toolResults] }); } else { this.appendMessage(errorMsg); this.#state.error = errorMessage; diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 2b5186010..3781ed431 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -43,6 +43,7 @@ import { V2_RETAINED_MESSAGE_TOKEN_BUDGET, } from "./compaction-v2-streaming"; import type { CompactionEntry, SessionEntry } from "./entries"; +import { isEstimateCacheable, readEstimateCache, writeEstimateCache } from "./message-cache"; import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages"; import { buildOpenAiNativeHistory, @@ -364,6 +365,21 @@ const IMAGE_TOKEN_ESTIMATE = 1200; * content) excludes them to avoid false triggers on thinking-heavy turns. */ export function estimateTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number { + // Settled historical messages are counted once and reused until an owner + // (prune/shake/strip-images) invalidates them; streaming assistants bypass + // the cache entirely (see message-cache.ts settle-gate invariant). + const cacheable = isEstimateCacheable(message); + const excludeEncryptedReasoning = options?.excludeEncryptedReasoning === true; + if (cacheable) { + const cached = readEstimateCache(message, excludeEncryptedReasoning); + if (cached !== undefined) return cached; + } + const result = computeMessageTokens(message, options); + if (cacheable) writeEstimateCache(message, excludeEncryptedReasoning, result); + return result; +} + +function computeMessageTokens(message: AgentMessage, options?: { excludeEncryptedReasoning?: boolean }): number { const fragments: string[] = []; let extra = 0; if ((message as { role?: string }).role === "bashExecution") { diff --git a/packages/agent/src/compaction/index.ts b/packages/agent/src/compaction/index.ts index 401215724..c0586f18d 100644 --- a/packages/agent/src/compaction/index.ts +++ b/packages/agent/src/compaction/index.ts @@ -6,6 +6,7 @@ export * from "./branch-summarization"; export * from "./compaction"; export * from "./entries"; export * from "./errors"; +export * from "./message-cache"; export * from "./messages"; export * from "./openai"; export * from "./pruning"; diff --git a/packages/agent/src/compaction/message-cache.ts b/packages/agent/src/compaction/message-cache.ts new file mode 100644 index 000000000..a2249f926 --- /dev/null +++ b/packages/agent/src/compaction/message-cache.ts @@ -0,0 +1,92 @@ +/** + * Per-message memoization for the two hot history walks: token estimation + * ({@link estimateTokens}) and LLM conversion (the coding-agent's `convertToLlm`). + * + * Long sessions re-walk a settled `AgentMessage[]` every turn, re-tokenizing and + * re-converting historical objects that only the newest suffix can change. These + * caches key on message *identity* so a settled message is counted/converted once + * and reused until an owner rewrites it. + * + * Correctness rests on two invariants: + * + * 1. **Settle gate.** A streaming assistant is mutated under one identity while + * its `usage`/`stopReason` are provisional (the seed carries zeroed usage and + * a placeholder `stopReason`). Caching it would freeze a mid-stream count, so + * estimation only caches assistants that are settled — real `usage` + * (`totalTokens > 0`) with a terminal `stopReason` that is not `"aborted"` / + * `"error"`. Unsettled assistants never read or insert. Non-assistant roles + * are immutable once appended and cache by identity. + * 2. **Owner invalidation.** `pruneToolOutputs` / `pruneSupersededToolResults`, + * `applyShakeRegion`, and `stripImagesFromMessage` rewrite message content in + * place under a stable identity. Each MUST call {@link invalidateMessageCache} + * on the mutated message before the next convert/estimate pass so both caches + * drop the stale entry. The convert cache lives in another package, so it + * subscribes via {@link registerMessageCacheInvalidator}. + */ +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; +import type { AgentMessage } from "../types"; + +/** External cache invalidators (e.g. the coding-agent `convertToLlm` memo). */ +const externalInvalidators = new Set<(message: AgentMessage) => void>(); + +/** + * Register a cache tied to message identity so owner mutations in this package + * (prune/shake) can invalidate it across the package boundary. Returns an + * unregister function. The coding-agent `convertToLlm` memo registers here. + */ +export function registerMessageCacheInvalidator(invalidate: (message: AgentMessage) => void): () => void { + externalInvalidators.add(invalidate); + return () => { + externalInvalidators.delete(invalidate); + }; +} + +// Dual option-split estimate caches: the compaction floor passes +// `excludeEncryptedReasoning` (dropping opaque provider reasoning), so a message +// has two distinct estimates that must not collide in one map. +// +// These are WeakMaps, not symbol-tagged properties, deliberately: callers spread +// messages to derive throwaway variants for counting — `estimateBranchSummaryTokens` +// does `estimateTokens({ ...message, content: truncated })`. A symbol-keyed cache +// value rides along an object spread, so the truncated clone would inherit (and +// return) the full-content estimate. Keying strictly on identity keeps the cache +// off spread copies, which get their own fresh count. +const estimateCacheDefault = new WeakMap(); +const estimateCacheFloored = new WeakMap(); + +/** + * True when this message's estimate is safe to cache by identity. Non-assistants + * are immutable once appended; assistants are cached only once settled (see the + * settle-gate invariant above). + */ +export function isEstimateCacheable(message: AgentMessage): boolean { + if (message.role !== "assistant") return true; + const assistant = message as AssistantMessage; + return ( + assistant.stopReason !== "aborted" && + assistant.stopReason !== "error" && + assistant.usage != null && + assistant.usage.totalTokens > 0 + ); +} + +/** Read a cached estimate for the given option split, or `undefined` on miss. */ +export function readEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean): number | undefined { + return (excludeEncryptedReasoning ? estimateCacheFloored : estimateCacheDefault).get(message); +} + +/** Store an estimate for the given option split. */ +export function writeEstimateCache(message: AgentMessage, excludeEncryptedReasoning: boolean, value: number): void { + (excludeEncryptedReasoning ? estimateCacheFloored : estimateCacheDefault).set(message, value); +} + +/** + * Drop every cached derivation of `message` after an in-place rewrite. Owners of + * mutation (prune, shake, strip-images) call this at the mutation seam so the + * next convert/estimate pass recomputes from the new content. + */ +export function invalidateMessageCache(message: AgentMessage): void { + estimateCacheDefault.delete(message); + estimateCacheFloored.delete(message); + for (const invalidate of externalInvalidators) invalidate(message); +} diff --git a/packages/agent/src/compaction/pruning.ts b/packages/agent/src/compaction/pruning.ts index 1ba9a7ac0..83d5c2dd5 100644 --- a/packages/agent/src/compaction/pruning.ts +++ b/packages/agent/src/compaction/pruning.ts @@ -6,6 +6,7 @@ import type { ToolResultMessage } from "@oh-my-pi/pi-ai"; import type { AgentMessage, AgentToolCall } from "../types"; import { estimateTokens } from "./compaction"; import type { SessionEntry, SessionMessageEntry } from "./entries"; +import { invalidateMessageCache } from "./message-cache"; import { collectToolCallsById, isProtectedToolResult, @@ -295,6 +296,7 @@ export function pruneSupersededToolResults(entries: SessionEntry[], config: Supe for (const candidate of toPrune) { candidate.message.content = [{ type: "text", text: candidate.notice }]; candidate.message.prunedAt = prunedAt; + invalidateMessageCache(candidate.message as AgentMessage); tokensSaved += estimatePrunedSavings(candidate.tokens, candidate.notice); } return { prunedCount: toPrune.length, tokensSaved }; @@ -398,6 +400,7 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = : createPrunedNotice(candidate.tokens); message.content = [{ type: "text", text: notice }]; message.prunedAt = prunedAt; + invalidateMessageCache(message as AgentMessage); prunedCount++; } diff --git a/packages/agent/src/compaction/shake.ts b/packages/agent/src/compaction/shake.ts index 7e0f2c5ed..db95486de 100644 --- a/packages/agent/src/compaction/shake.ts +++ b/packages/agent/src/compaction/shake.ts @@ -15,6 +15,7 @@ import { countTokens } from "../tokenizer"; import type { AgentMessage } from "../types"; import { estimateTokens } from "./compaction"; import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries"; +import { invalidateMessageCache } from "./message-cache"; import { collectToolCallsById, isProtectedToolResult, @@ -406,12 +407,18 @@ export function applyShakeRegion(region: ShakeRegion, replacement: string): void const message = region.entry.message as ToolResultMessage; message.content = [{ type: "text", text: replacement }]; message.prunedAt = Date.now(); + invalidateMessageCache(message as AgentMessage); return; } const slot = getBlockTextSlot(region.entry, region.blockIndex); if (!slot) return; const text = slot.read(); slot.write(text.slice(0, region.start) + replacement + text.slice(region.end)); + // Message entries keep a stable `entry.message` identity across context + // rebuilds, so an in-place block rewrite must drop its cached estimate/convert. + // Custom-message entries are re-materialized into a fresh AgentMessage on every + // buildSessionContext, so they carry no stable cached identity to invalidate. + if (region.entry.type === "message") invalidateMessageCache(region.entry.message); } /** diff --git a/packages/agent/src/proxy.ts b/packages/agent/src/proxy.ts index b3043b659..f96d0b766 100644 --- a/packages/agent/src/proxy.ts +++ b/packages/agent/src/proxy.ts @@ -8,6 +8,7 @@ import { type Context, EventStream, type FetchImpl, + type ImageContent, type Model, type SimpleStreamOptions, type StopReason, @@ -47,6 +48,7 @@ export type ProxyAssistantMessageEvent = | { type: "thinking_start"; contentIndex: number } | { type: "thinking_delta"; contentIndex: number; delta: string } | { type: "thinking_end"; contentIndex: number; contentSignature?: string } + | { type: "image_end"; contentIndex: number; content: ImageContent } | { type: "toolcall_start"; contentIndex: number; id: string; toolName: string } | { type: "toolcall_delta"; contentIndex: number; delta: string } | { type: "toolcall_end"; contentIndex: number } @@ -315,6 +317,15 @@ function processProxyEvent( throw new Error("Received thinking_end for non-thinking content"); } + case "image_end": + partial.content[proxyEvent.contentIndex] = proxyEvent.content; + return { + type: "image_end", + contentIndex: proxyEvent.contentIndex, + content: proxyEvent.content, + partial, + }; + case "toolcall_start": partial.content[proxyEvent.contentIndex] = { type: "toolCall", diff --git a/packages/agent/src/telemetry.ts b/packages/agent/src/telemetry.ts index e6314a325..55ca9a82a 100644 --- a/packages/agent/src/telemetry.ts +++ b/packages/agent/src/telemetry.ts @@ -957,6 +957,9 @@ function assistantContentToOtelParts(content: AssistantMessage["content"]): Otel case "text": parts.push({ type: "text", content: part.text }); break; + case "image": + parts.push({ type: "blob", modality: "image", mime_type: part.mimeType, content: part.data }); + break; case "thinking": parts.push({ type: "reasoning", content: part.thinking }); break; diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index 5119d38c2..9fe6078ab 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -657,14 +657,16 @@ export interface AgentTool>) => boolean); /** * Controls how the INTENT_FIELD (`i`) is handled for this tool. * - `"require"` (default): `i` is injected and required in the parameter schema. diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index 64edff64f..9cc19154e 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -630,7 +630,7 @@ describe("agentLoop with AgentMessage", () => { expect(finalTurn.content).toContainEqual({ type: "text", text: "done after recovery" }); }); - it("does not recover completed tool calls after non-stream transient errors", async () => { + it("runs completed tool calls after a transient stream JSON parse error", async () => { const executedParams: Array<{ value: string }> = []; const toolSchema = type({ value: "string" }); const tool: AgentTool = { @@ -652,9 +652,9 @@ describe("agentLoop with AgentMessage", () => { { content: [{ type: "toolCall", id: "tool-1", name: "echo", arguments: { value: "hello" } }], stopReason: "error", - errorMessage: "rate_limit_error", + errorMessage: "JSON Parse error: Unterminated string", }, - { content: ["should not continue"] }, + { content: ["done after parse recovery"] }, ], }); const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; @@ -667,12 +667,275 @@ describe("agentLoop with AgentMessage", () => { mock.stream, ).result(); - expect(executedParams).toEqual([]); + expect(executedParams).toEqual([{ value: "hello" }]); + expect(mock.calls).toHaveLength(2); + expect(messages.map(message => message.role)).toEqual(["user", "assistant", "toolResult", "assistant"]); + const recoveredTurn = messages[1] as AssistantMessage; + expect(recoveredTurn.stopReason).toBe("toolUse"); + expect(recoveredTurn.stopDetails?.type).toBe("stream_interrupted_after_content"); + const finalTurn = messages[3] as AssistantMessage; + expect(finalTurn.content).toContainEqual({ type: "text", text: "done after parse recovery" }); + }); + + it("recovers only completed calls when a stream parse error interrupts the next call", async () => { + const executedParams: Array<{ value: string }> = []; + const toolSchema = type({ value: "string" }); + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + async execute(_toolCallId, params) { + executedParams.push(params); + return { + content: [{ type: "text", text: `echoed: ${params.value}` }], + details: { value: params.value }, + }; + }, + }; + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + const completedCall = { + type: "toolCall" as const, + id: "tool-complete", + name: "echo", + arguments: { value: "complete" }, + }; + const incompleteCall = { + type: "toolCall" as const, + id: "tool-incomplete", + name: "echo", + arguments: { value: "incomplete" }, + }; + const mock = createMockModel({ responses: [{ content: ["done after partial parse recovery"] }] }); + let streamCalls = 0; + const streamFn: typeof mock.stream = (model, callContext, options) => { + streamCalls++; + if (streamCalls > 1) return mock.stream(model, callContext, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + const partial: AssistantMessage = { + ...createAssistantMessage([completedCall, incompleteCall], "error"), + errorMessage: "provider stream parse failed", + stopDetails: { + type: "parse_error", + explanation: "JSON Parse error: Unterminated string", + }, + }; + stream.push({ type: "start", partial }); + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: completedCall, partial }); + stream.push({ type: "toolcall_start", contentIndex: 1, partial }); + stream.push({ type: "toolcall_delta", contentIndex: 1, delta: '{"value":"incomplete', partial }); + stream.push({ type: "error", reason: "error", error: partial }); + }); + return stream; + }; + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; + + const messages = await agentLoop([createUserMessage("run echo")], context, config, undefined, streamFn).result(); + + expect(executedParams).toEqual([{ value: "complete" }]); + expect(streamCalls).toBe(2); expect(mock.calls).toHaveLength(1); - expect(messages.map(message => message.role)).toEqual(["user", "assistant", "toolResult"]); + expect(messages.map(message => message.role)).toEqual(["user", "assistant", "toolResult", "assistant"]); + const recoveredTurn = messages[1] as AssistantMessage; + expect(recoveredTurn.stopReason).toBe("toolUse"); + expect(recoveredTurn.content.filter(block => block.type === "toolCall").map(block => block.id)).toEqual([ + "tool-complete", + ]); + expect(messages.some(message => message.role === "toolResult" && message.toolCallId === "tool-incomplete")).toBe( + false, + ); + }); + + it("does not recover a wrapped refusal after dropping an incomplete sibling", async () => { + const executedParams: Array<{ value: string }> = []; + const toolSchema = type({ value: "string" }); + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + async execute(_toolCallId, params) { + executedParams.push(params); + return { content: [], details: { value: params.value } }; + }, + }; + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + const completedCall = { + type: "toolCall" as const, + id: "tool-complete", + name: "echo", + arguments: { value: "complete" }, + }; + const incompleteCall = { + type: "toolCall" as const, + id: "tool-incomplete", + name: "echo", + arguments: { value: "incomplete" }, + }; + const mock = createMockModel({ responses: [{ content: ["should not continue"] }] }); + let streamCalls = 0; + const streamFn: typeof mock.stream = (model, callContext, options) => { + streamCalls++; + if (streamCalls > 1) return mock.stream(model, callContext, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + const partial: AssistantMessage = { + ...createAssistantMessage([completedCall, incompleteCall], "error"), + errorMessage: "provider refused output", + stopDetails: { type: "refusal", explanation: "Unexpected end of JSON input" }, + }; + stream.push({ type: "start", partial }); + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: completedCall, partial }); + stream.push({ type: "toolcall_start", contentIndex: 1, partial }); + stream.push({ type: "toolcall_delta", contentIndex: 1, delta: '{"value":"incomplete', partial }); + stream.push({ type: "error", reason: "error", error: partial }); + }); + return stream; + }; + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; + + const messages = await agentLoop([createUserMessage("run echo")], context, config, undefined, streamFn).result(); + + expect(executedParams).toEqual([]); + expect(streamCalls).toBe(1); + expect(mock.calls).toHaveLength(0); const errorTurn = messages[1] as AssistantMessage; expect(errorTurn.stopReason).toBe("error"); - expect(errorTurn.errorMessage).toBe("rate_limit_error"); + expect(errorTurn.content.filter(block => block.type === "toolCall").map(block => block.id)).toEqual([ + "tool-complete", + ]); + expect(errorTurn.stopDetails).toEqual({ + type: "stream_interrupted_after_content", + category: "refusal", + explanation: "Unexpected end of JSON input", + }); + }); + + it("does not recover a mixed known and unknown completed tool turn", async () => { + const executedParams: Array<{ value: string }> = []; + const toolSchema = type({ value: "string" }); + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + async execute(_toolCallId, params) { + executedParams.push(params); + return { content: [], details: { value: params.value } }; + }, + }; + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + const mock = createMockModel({ + responses: [ + { + content: [ + { type: "toolCall", id: "tool-known", name: "echo", arguments: { value: "known" } }, + { type: "toolCall", id: "tool-unknown", name: "missing", arguments: { value: "unknown" } }, + ], + stopReason: "error", + errorMessage: "JSON Parse error: Unterminated string", + }, + { content: ["should not continue"] }, + ], + }); + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; + + const messages = await agentLoop( + [createUserMessage("run mixed tools")], + context, + config, + undefined, + mock.stream, + ).result(); + + expect(executedParams).toEqual([]); + expect(mock.calls).toHaveLength(1); + const errorTurn = messages[1] as AssistantMessage; + expect(errorTurn.stopReason).toBe("error"); + expect(errorTurn.errorMessage).toBe("JSON Parse error: Unterminated string"); + }); + + it("does not recover content-only transient stream parse errors", async () => { + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [] }; + const mock = createMockModel({ + responses: [ + { + content: ["partial response"], + stopReason: "error", + errorMessage: "JSON Parse error: Unterminated string", + }, + { content: ["should not continue"] }, + ], + }); + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; + + const messages = await agentLoop( + [createUserMessage("answer once")], + context, + config, + undefined, + mock.stream, + ).result(); + + expect(mock.calls).toHaveLength(1); + expect(messages.map(message => message.role)).toEqual(["user", "assistant"]); + const errorTurn = messages[1] as AssistantMessage; + expect(errorTurn.stopReason).toBe("error"); + expect(errorTurn.errorMessage).toBe("JSON Parse error: Unterminated string"); + }); + + it("does not recover ordinary transient errors or terminal stops quoting parse diagnostics", async () => { + for (const stopDetails of [ + undefined, + { type: "refusal", explanation: "Unexpected end of JSON input" }, + { type: "sensitive", explanation: "Unexpected end of JSON input" }, + ]) { + const executedParams: Array<{ value: string }> = []; + const toolSchema = type({ value: "string" }); + const tool: AgentTool = { + name: "echo", + label: "Echo", + description: "Echo tool", + parameters: toolSchema, + async execute(_toolCallId, params) { + executedParams.push(params); + return { + content: [{ type: "text", text: `echoed: ${params.value}` }], + details: { value: params.value }, + }; + }, + }; + const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; + const mock = createMockModel({ + responses: [ + { + content: [{ type: "toolCall", id: "tool-1", name: "echo", arguments: { value: "hello" } }], + stopReason: "error", + stopDetails, + errorMessage: "rate_limit_error", + }, + { content: ["should not continue"] }, + ], + }); + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; + + const messages = await agentLoop( + [createUserMessage("run echo")], + context, + config, + undefined, + mock.stream, + ).result(); + + expect(executedParams).toEqual([]); + expect(mock.calls).toHaveLength(1); + expect(messages.map(message => message.role)).toEqual(["user", "assistant", "toolResult"]); + const errorTurn = messages[1] as AssistantMessage; + expect(errorTurn.stopReason).toBe("error"); + expect(errorTurn.errorMessage).toBe("rate_limit_error"); + expect(errorTurn.stopDetails).toEqual(stopDetails); + } }); it("labels the synthetic tool result for a provider-error turn as not-executed and preserves the upstream error", async () => { @@ -1691,8 +1954,8 @@ describe("agentLoop with AgentMessage", () => { } }); - it("does not abort a non-interruptible tool mid-wait; steering still drains at the boundary", async () => { - const toolSchema = type({}); + it("does not abort a tool when its interruptibility resolver rejects the call", async () => { + const toolSchema = type({ op: "'start' | 'wait'" }); let steerReady = false; let drained = false; let observedAbort = false; @@ -1701,8 +1964,9 @@ describe("agentLoop with AgentMessage", () => { const tool: AgentTool> = { name: "wait", label: "Wait", - description: "Blocks on its own window (no interruptible flag)", + description: "Blocks on its own window (mimics a side-effecting start)", parameters: toolSchema, + interruptible: params => params.op === "wait", async execute(_toolCallId, _params, signal) { steerReady = true; const { promise, resolve } = Promise.withResolvers(); @@ -1731,7 +1995,7 @@ describe("agentLoop with AgentMessage", () => { const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] }; const mock = createMockModel({ responses: [ - { content: [{ type: "toolCall", id: "tool-1", name: "wait", arguments: {} }] }, + { content: [{ type: "toolCall", id: "tool-1", name: "wait", arguments: { op: "start" } }] }, { content: ["done"] }, ], }); diff --git a/packages/agent/test/agent-side-request-context.test.ts b/packages/agent/test/agent-side-request-context.test.ts index 3c38d5652..20299e34b 100644 --- a/packages/agent/test/agent-side-request-context.test.ts +++ b/packages/agent/test/agent-side-request-context.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, mock } from "bun:test"; import { type AssistantMessage, type Context, z } from "@oh-my-pi/pi-ai"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Agent } from "../src/agent"; import type { AgentTool } from "../src/types"; @@ -39,6 +40,19 @@ function testAssistantMessage(text: string): AssistantMessage { }; } +const cursorModel = buildModel({ + id: "cursor-test", + name: "Cursor Test", + api: "cursor-agent", + provider: "cursor", + baseUrl: "https://example.invalid", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 8_192, + maxTokens: 2_048, +}); + describe("Agent — buildSideRequestContext", () => { const model = createMockModel({ responses: [] }); const tool: AgentTool = { @@ -109,6 +123,43 @@ describe("Agent — buildSideRequestContext", () => { }); }); + it("adds mounted Cursor tools to main and side provider contexts", async () => { + await withNativeDialectEnv(async () => { + const mountedTool: AgentTool = { + ...tool, + name: "mcp__fixture_report", + label: "Fixture Report", + }; + let mainContext: Context | undefined; + const agent = new Agent({ + initialState: { + model: cursorModel, + systemPrompt: ["system"], + tools: [tool], + }, + getCursorTools: () => [tool, mountedTool], + streamFn: (_model, context) => { + mainContext = context; + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + const message = testAssistantMessage("ok"); + stream.push({ type: "text_delta", contentIndex: 0, delta: "ok", partial: message }); + stream.push({ type: "done", reason: "stop", message }); + }); + return stream; + }, + }); + + await agent.prompt("Q?"); + const sideContext = await agent.buildSideRequestContext([ + { role: "user", content: [{ type: "text", text: "Q?" }], timestamp: Date.now() }, + ]); + + expect(mainContext?.tools?.map(entry => entry.name)).toEqual(["test_tool", "mcp__fixture_report"]); + expect(sideContext.tools?.map(entry => entry.name)).toEqual(["test_tool", "mcp__fixture_report"]); + }); + }); + it("returns empty tools when owned dialect is active", async () => { const agent = new Agent({ initialState: { diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index 178ff84be..30e24f1b4 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -1,7 +1,8 @@ import { describe, expect, it } from "bun:test"; import { Agent, type AgentEvent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; -import { type SimpleStreamOptions, z } from "@oh-my-pi/pi-ai"; +import { type SimpleStreamOptions, type ToolResultMessage, z } from "@oh-my-pi/pi-ai"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; +import { kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { createAssistantMessage } from "./helpers"; @@ -184,7 +185,7 @@ describe("Agent", () => { expect(lastMessage.errorMessage).toBe(errorText); }); - it("prompt() keeps unrelated provider stream failures out of the assistant lifecycle", async () => { + it("prompt() emits assistant error lifecycle for provider stream failures", async () => { const mock = createMockModel({ responses: [] }); const errorText = "connection reset"; const agent = new Agent({ @@ -201,17 +202,138 @@ describe("Agent", () => { await agent.prompt("trigger"); unsubscribe(); - expect(events.some(event => event.type === "message_start" && event.message.role === "assistant")).toBe(false); - expect(events.some(event => event.type === "message_end" && event.message.role === "assistant")).toBe(false); - const agentEnd = events.find(event => event.type === "agent_end"); - if (agentEnd?.type !== "agent_end") { - throw new Error("agent_end not emitted"); + const assistantStartIndex = events.findIndex( + event => event.type === "message_start" && event.message.role === "assistant", + ); + const assistantEndIndex = events.findIndex( + event => event.type === "message_end" && event.message.role === "assistant", + ); + const turnEndIndex = events.findIndex(event => event.type === "turn_end"); + const agentEndIndex = events.findIndex(event => event.type === "agent_end"); + expect(assistantStartIndex).toBeGreaterThan(-1); + expect(assistantEndIndex).toBeGreaterThan(assistantStartIndex); + expect(turnEndIndex).toBeGreaterThan(assistantEndIndex); + expect(agentEndIndex).toBeGreaterThan(turnEndIndex); + + const assistantEnd = events[assistantEndIndex]; + if (assistantEnd?.type !== "message_end" || assistantEnd.message.role !== "assistant") { + throw new Error("assistant message_end not emitted"); } - const errorMessage = agentEnd.messages.find(message => message.role === "assistant"); - if (errorMessage?.role !== "assistant") { - throw new Error("assistant error was not included in agent_end"); - } - expect(errorMessage.errorMessage).toBe(errorText); + expect(assistantEnd.message.stopReason).toBe("error"); + expect(assistantEnd.message.errorMessage).toBe(errorText); + }); + + it("pairs tool calls from failed partial streams with synthetic tool results", async () => { + const mock = createMockModel({ responses: [] }); + const errorText = "connection reset after tool call"; + const toolCall = { type: "toolCall" as const, id: "tool-1", name: "alpha", arguments: { value: "hello" } }; + const started = createAssistantMessage([toolCall]); + const agent = new Agent({ + initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, + streamFn: () => { + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + stream.push({ type: "start", partial: started }); + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall, partial: started }); + stream.fail(new Error(errorText)); + }); + return stream; + }, + }); + const events: AgentEvent[] = []; + const unsubscribe = agent.subscribe(event => events.push(event)); + + await agent.prompt("trigger"); + unsubscribe(); + + const toolResult = agent.state.messages.find(message => message.role === "toolResult"); + expect(toolResult).toMatchObject({ + role: "toolResult", + toolCallId: "tool-1", + toolName: "alpha", + isError: true, + details: { + __synthetic: true, + source: "assistant_stop_error", + executed: false, + upstreamError: errorText, + }, + }); + + const turnEnd = events.find(event => event.type === "turn_end"); + expect(turnEnd).toMatchObject({ + type: "turn_end", + toolResults: [{ role: "toolResult", toolCallId: "tool-1", isError: true }], + }); + }); + + it("drops incomplete tool calls when a partial stream fails before toolcall_end", async () => { + const mock = createMockModel({ responses: [] }); + const started = createAssistantMessage([{ type: "toolCall", id: "tool-1", name: "alpha", arguments: {} }]); + const agent = new Agent({ + initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, + streamFn: () => { + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + stream.push({ type: "start", partial: started }); + stream.push({ type: "toolcall_start", contentIndex: 0, partial: started }); + stream.push({ type: "toolcall_delta", contentIndex: 0, delta: '{"value":', partial: started }); + stream.fail(new Error("connection reset during tool arguments")); + }); + return stream; + }, + }); + + await agent.prompt("trigger"); + + const assistant = agent.state.messages.find(message => message.role === "assistant"); + expect(assistant?.content.some(block => block.type === "toolCall")).toBe(false); + expect(agent.state.messages.some(message => message.role === "toolResult")).toBe(false); + }); + + it("preserves buffered Cursor results when a partial stream fails", async () => { + const mock = createMockModel({ responses: [] }); + const errorText = "connection reset after Cursor exec"; + const toolCall = { + type: "toolCall" as const, + id: "cursor-tool-1", + name: "shell", + arguments: { command: "pwd" }, + [kCursorExecResolved]: true, + }; + const started = createAssistantMessage([toolCall]); + const realToolResult: ToolResultMessage = { + role: "toolResult", + toolCallId: toolCall.id, + toolName: toolCall.name, + content: [{ type: "text", text: "/workspace" }], + isError: false, + timestamp: Date.now(), + }; + const agent = new Agent({ + initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, + cursorOnToolResult: message => message, + streamFn: (_model, _context, options) => { + const stream = new AssistantMessageEventStream(); + queueMicrotask(async () => { + await options?.cursorOnToolResult?.(realToolResult); + stream.push({ type: "start", partial: started }); + stream.fail(new Error(errorText)); + }); + return stream; + }, + }); + + await agent.prompt("trigger"); + + const toolResults = agent.state.messages.filter(message => message.role === "toolResult"); + expect(toolResults).toHaveLength(1); + expect(toolResults[0]).toMatchObject({ + toolCallId: toolCall.id, + toolName: toolCall.name, + content: [{ type: "text", text: "/workspace" }], + isError: false, + }); }); it("prompt() finalizes an existing assistant stream for Anthropic output-blocked stream errors", async () => { diff --git a/packages/agent/test/message-cache.test.ts b/packages/agent/test/message-cache.test.ts new file mode 100644 index 000000000..ac3db9997 --- /dev/null +++ b/packages/agent/test/message-cache.test.ts @@ -0,0 +1,159 @@ +import { describe, expect, test } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { SessionMessageEntry } from "@oh-my-pi/pi-agent-core/compaction"; +import { + applyShakeRegion, + collectShakeRegions, + DEFAULT_PRUNE_CONFIG, + estimateTokens, + invalidateMessageCache, + isEstimateCacheable, + pruneToolOutputs, +} from "@oh-my-pi/pi-agent-core/compaction"; +import type { AssistantMessage, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai"; + +let idCounter = 0; +function nextId(): string { + return `mc-${idCounter++}`; +} + +function messageEntry(message: AgentMessage): SessionMessageEntry { + return { type: "message", id: nextId(), parentId: null, timestamp: new Date().toISOString(), message }; +} + +function usage(totalTokens: number): Usage { + return { + input: totalTokens, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; +} + +function settledAssistant(text: string): AssistantMessage { + return { + role: "assistant", + content: [{ type: "text", text }], + api: "anthropic-messages", + provider: "anthropic", + model: "bench", + usage: usage(120), + stopReason: "stop", + timestamp: 1, + }; +} + +function toolResult(text: string, extra?: Partial): ToolResultMessage { + return { + role: "toolResult", + toolCallId: `call-${idCounter++}`, + toolName: "read", + content: [{ type: "text", text }], + isError: false, + timestamp: Date.now(), + ...extra, + }; +} + +describe("estimate cache settle gate", () => { + test("caches settled assistants (terminal stopReason + real usage)", () => { + expect(isEstimateCacheable(settledAssistant("done"))).toBe(true); + }); + + test("bypasses a streaming assistant (zero usage seed)", () => { + const streaming: AssistantMessage = { ...settledAssistant("partial"), usage: usage(0), stopReason: "stop" }; + expect(isEstimateCacheable(streaming)).toBe(false); + }); + + test("bypasses aborted and error assistants even with usage", () => { + expect(isEstimateCacheable({ ...settledAssistant("x"), stopReason: "aborted" })).toBe(false); + expect(isEstimateCacheable({ ...settledAssistant("x"), stopReason: "error" })).toBe(false); + }); + + test("caches non-assistant roles unconditionally", () => { + expect(isEstimateCacheable(toolResult("out") as AgentMessage)).toBe(true); + expect(isEstimateCacheable({ role: "user", content: "hi", timestamp: 1 } as AgentMessage)).toBe(true); + }); + + test("a streaming assistant re-estimates as its content grows", () => { + const streaming: AssistantMessage = { + ...settledAssistant("first chunk"), + usage: usage(0), + stopReason: "stop", + }; + const before = estimateTokens(streaming as AgentMessage); + streaming.content = [{ type: "text", text: "first chunk plus a much longer continuation of streamed text" }]; + const after = estimateTokens(streaming as AgentMessage); + // Unsettled assistants never read the cache, so the grown content is recounted. + expect(after).toBeGreaterThan(before); + }); +}); + +describe("estimate cache option split", () => { + test("default and floored estimates do not collide in one map", () => { + const blob = "blob ".repeat(4000); + const msg: AssistantMessage = { + ...settledAssistant("thinking heavy"), + content: [ + { type: "text", text: "answer" }, + { type: "thinking", thinking: "reasoning", thinkingSignature: blob }, + ], + }; + // Prime the default map first, then the floored one; the floored estimate + // (which drops the encrypted-reasoning blob) must not read the default entry. + const withBlob = estimateTokens(msg as AgentMessage); + const floored = estimateTokens(msg as AgentMessage, { excludeEncryptedReasoning: true }); + expect(withBlob).toBeGreaterThan(floored + 500); + // Cached reads return the same split values. + expect(estimateTokens(msg as AgentMessage)).toBe(withBlob); + expect(estimateTokens(msg as AgentMessage, { excludeEncryptedReasoning: true })).toBe(floored); + }); +}); + +describe("estimate cache invalidation seams", () => { + test("pruneToolOutputs drops the cached estimate of a pruned result", () => { + const big = toolResult("x".repeat(20_000)); + const entries = [messageEntry(big as AgentMessage)]; + const before = estimateTokens(big as AgentMessage); + expect(before).toBeGreaterThan(1000); + + const result = pruneToolOutputs(entries, { ...DEFAULT_PRUNE_CONFIG, protectTokens: 0, minimumSavings: 0 }); + expect(result.prunedCount).toBe(1); + + // After the in-place prune the estimate must reflect the short placeholder, + // not the stale full-content count. + const after = estimateTokens(big as AgentMessage); + expect(after).toBeLessThan(before); + }); + + test("applyShakeRegion drops the cached estimate of a shaken result", () => { + const big = toolResult(`\`\`\`ts\n${"const value = compute(a, b, c, d, e);\n".repeat(400)}\`\`\``); + const entry = messageEntry(big as AgentMessage); + const before = estimateTokens(big as AgentMessage); + + const regions = collectShakeRegions([entry], { + protectTokens: 0, + minSavings: 0, + protectedTools: [], + fenceMinTokens: 0, + }); + expect(regions.length).toBeGreaterThan(0); + applyShakeRegion(regions[0], "[shaken]"); + + const after = estimateTokens(big as AgentMessage); + expect(after).toBeLessThan(before); + }); + + test("explicit invalidateMessageCache forces a recount", () => { + const result = toolResult("original content here"); + const before = estimateTokens(result as AgentMessage); + // Mutate content directly (simulating an owner rewrite) then invalidate. + result.content = [{ type: "text", text: "a much longer replacement body that should count higher than before" }]; + // Without invalidation the stale cached value would still be returned. + expect(estimateTokens(result as AgentMessage)).toBe(before); + invalidateMessageCache(result as AgentMessage); + expect(estimateTokens(result as AgentMessage)).toBeGreaterThan(before); + }); +}); diff --git a/packages/agent/test/pause-gate.test.ts b/packages/agent/test/pause-gate.test.ts index 5e20cc030..1ff94c72c 100644 --- a/packages/agent/test/pause-gate.test.ts +++ b/packages/agent/test/pause-gate.test.ts @@ -51,6 +51,16 @@ describe("agentPauseGate", () => { it("holds tool execution at the tool boundary when paused mid-turn", async () => { const executed: string[] = []; + // Signal exactly when the loop parks on the gate. A test-local manual + // patch (not vi.spyOn) so a sibling file's restoreAllMocks cannot remove + // it, and a gate regression that never parks hangs this await (test + // timeout) instead of racing past a vacuous assertion. + const toolBoundary = Promise.withResolvers(); + const originalWait = agentPauseGate.waitUntilResumed; + agentPauseGate.waitUntilResumed = (signal?: AbortSignal) => { + toolBoundary.resolve(); + return originalWait.call(agentPauseGate, signal); + }; const mock = createMockModel({ responses: [ () => { @@ -65,15 +75,19 @@ describe("agentPauseGate", () => { const context: AgentContext = { systemPrompt: ["Test"], messages: [], tools: [makeEchoTool(executed)] }; const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; - const result = agentLoop([createUserMessage("run echo")], context, config, undefined, mock.stream).result(); - await Bun.sleep(20); - expect(executed).toEqual([]); // tool parked, not started - expect(mock.calls.length).toBe(1); // and no follow-up model call either + try { + const result = agentLoop([createUserMessage("run echo")], context, config, undefined, mock.stream).result(); + await toolBoundary.promise; + expect(executed).toEqual([]); // tool parked, not started + expect(mock.calls.length).toBe(1); // and no follow-up model call either - agentPauseGate.resume(); - await result; - expect(executed).toEqual(["frozen"]); - expect(mock.calls.length).toBe(2); + agentPauseGate.resume(); + await result; + expect(executed).toEqual(["frozen"]); + expect(mock.calls.length).toBe(2); + } finally { + agentPauseGate.waitUntilResumed = originalWait; + } }); it("lets an external abort unwind a parked run without releasing the gate", async () => { diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d3095bf01..b90401767 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,102 @@ ## [Unreleased] +### Added + +- Added Synthetic (synthetic.new) usage provider: `/usage` now reports the rolling 5-hour request limit and weekly credit quota via `GET /v2/quotas`, including per-tick regeneration rates in the window labels. +- Added optional `UsageWindow.resetLabel` so rolling windows can render their countdown with an accurate verb (e.g. "tick in 12m" / "regen in 51m" instead of "resets in") — both quota windows on Synthetic regenerate incrementally rather than hard-resetting. + +### Fixed + +- Fixed GitHub Copilot OpenAI-compatible requests being rejected when the session's native OpenAI service tier was set to `priority` ([#5160](https://github.com/can1357/oh-my-pi/pull/5160) by [@audreyt](https://github.com/audreyt)). +- Fixed OpenAI Responses token-cap truncations suppressing fully streamed function and custom tool calls whose inputs are complete. +- Added SuperGrok (`xai-oauth`) usage tracking for weekly credits, product limits, and positive on-demand caps. + +## [17.0.8] - 2026-07-22 + +### Fixed + +- Fixed Gemini Flash Cloud Code Assist empty-response retries when responses contain only intercepted planning-leak JSON. +- Fixed Antigravity auto-routing to correctly fail over to the sandbox endpoint when the daily endpoint exhausts its retries. +- Fixed OpenAI-compatible providers configured with auth: none incorrectly sending an Authorization: Bearer N/A header, which broke custom endpoints using alternative authentication headers. +- Fixed auth-gateway model listings exposing duplicate or ambiguous model IDs by ensuring only provider-qualified routing IDs are advertised. +- Improved connection error handling by classifying generic connection failures as transient, allowing them to be retried, while keeping explicit authentication rejections non-retryable. +- Fixed custom Anthropic base URLs losing native thinking signatures during continuation requests. +- Fixed Alibaba Coding Plan Custom login rejecting valid API keys on endpoints that do not serve the default validation model by validating against the model catalog instead. + +## [17.0.6] - 2026-07-20 + +### Fixed + +- Fixed OpenAI Codex credentials limited to one ChatGPT workspace per email: a personal Plus/Pro plan and a Team/Enterprise seat under the same email now coexist in the auth store — with separate rotation and usage pools — instead of the second login silently replacing the first. The workspace (`chatgpt_account_id`) is captured as the credential's org at login with the plan type as its display label, and two members of one workspace keep separate rows ([#2966](https://github.com/can1357/oh-my-pi/issues/2966)). +- Fixed Devin total-token usage omitting cache reads and cache writes. +- Fixed model switches to Devin rejecting foreign provider response IDs, reasoning signatures, and empty interrupted turns as invalid Cascade history. +- Classified zero-output Devin `invalid_argument` trailers as context overflow when the serialized message history is already large, routing cumulative tool-output payload failures through context maintenance—including artifact-backed shake rescue—instead of retrying the same rejected history. + +## [17.0.5] - 2026-07-18 + +### Changed + +- Changed Anthropic API-key requests to default to a 1-hour prompt-cache retention (using the extended-cache-ttl-2025-04-11 beta) to prevent cold-misses during idle sessions, with support for PI_CACHE_RETENTION values "short" and "none" to override this behavior. + +### Fixed + +- Fixed transient OpenAI stream truncations by retrying once before output becomes replay-unsafe, preventing recoverable transport errors from failing the turn. +- Fixed native Kimi Code K3 thinking being disabled during named function selection by utilizing generic required tool choice. +- Fixed /login moonshot validating China-platform API keys against the international host instead of honoring MOONSHOT_BASE_URL. +- Fixed Anthropic session stickiness suppressing usage-based re-ranking indefinitely by gating stickiness on a 1-hour cache warmth window (configurable via ANTHROPIC_SESSION_STICKY_CACHE_WARM_MS) to restore proactive multi-account load balancing after long idle periods. +- Fixed credential ranking where clockless Anthropic usage windows incorrectly outranked clocked sibling credentials. +- Fixed tool request failures (HTTP 400) on local grammar-constrained OpenAI-compatible backends (such as llama.cpp, LM Studio, and vLLM) by widening bare boolean subschemas into a value-accepting primitive union. +- Fixed custom OAuth Anthropic-compatible endpoints receiving generated Claude Code fingerprint headers even when explicit header overrides were provided. +- Fixed active sessions for plan-gated OpenAI Codex models (Sol/Luna) silently re-routing to sibling OAuth accounts when usage headroom changed, ensuring session stickiness is preserved as long as the preferred credential remains usable and eligible. + +## [17.0.4] - 2026-07-18 + +### Fixed + +- Fixed Kimi Code usage reports dropping the 5h window reset time (`omp usage` showed no "resets in …" for the 5h limit): the API returns `resetTime` on the limit `detail`, not on `window`, so the parsed row-level reset is now carried onto the window when the window itself has none. +- Made Kimi device-id persistence best-effort: a missing or unwritable `~/.omp/agent` directory no longer throws during Kimi header construction, which silently nulled every `kimi-code` usage probe on fresh installs. +- Coerced boolean tool-schema subschemas to MFJS object forms for native Moonshot/Kimi endpoints, preventing the task tool's `outputSchema` field from causing HTTP 400 responses ([#5952](https://github.com/can1357/oh-my-pi/issues/5952)). + +## [17.0.3] - 2026-07-17 + +### Fixed + +- Replaced the opaque `h2 is not supported` failure on the Cursor run transport with an actionable error naming the ALPN-stripping proxy as the cause and pointing at the `providers.cursor.baseUrl` HTTP/2 bridge workaround. The run RPC is HTTP/2-only, so behind a TLS-intercepting proxy that strips ALPN (e.g. Zscaler) bun cannot negotiate `h2` and the completion cannot proceed ([#5828](https://github.com/can1357/oh-my-pi/issues/5828)). +- Restored the `createAssistantMessageEventStream()` root export used by legacy provider extensions ([#5879](https://github.com/can1357/oh-my-pi/issues/5879)). +- Fixed parallel Responses tool-result images interleaving synthetic user messages before all pending outputs, preventing strict OpenRouter/Moonshot backends from rejecting follow-up requests. ([#5850](https://github.com/can1357/oh-my-pi/issues/5850)) +- Fixed Kimi Code K3 requests to send native named efforts (`low`, `high`, `max`) and use adaptive effort rather than generic token budgets on explicit Anthropic transport overrides ([#5893](https://github.com/can1357/oh-my-pi/issues/5893)). +- Automatically invalidate and rotate OAuth credentials when an "invalidated oauth token" error occurs +- Fixed Anthropic usage reports treating the organization response header as the account identity, which caused the 5h/7d status-line segment to disappear for OAuth credentials without stored organization metadata. ([#5698](https://github.com/can1357/oh-my-pi/issues/5698)) + +## [17.0.2] - 2026-07-17 + +### Fixed + +- Automatically invalidate and rotate OAuth credentials when an "invalidated oauth token" error occurs. +- Fixed auth-broker snapshot validation rejecting API keys stored via the `/login` flow, restoring support for gateway/broker setups serving login-sourced keys on custom hosts. +- Fixed an issue where literal reasoning tags (e.g., ``) inside Markdown code blocks or inline code were incorrectly treated as reasoning boundaries, which corrupted the rendered Markdown. +- Classified HTTP 402 and "balance exhausted" quota responses as persistent usage limits, enabling automatic rotation of multi-account requests to a sibling credential. +- Fixed `kimi-code` Anthropic-format requests ignoring custom provider base URLs. +- Fixed an issue where GPT-5.6 Codex Responses-Lite requests failed with an HTTP 400 error due to invalid `tool_choice` parameters after tools were rewritten, by automatically downgrading forced hosted choices to `tool_choice: "auto"` while preserving explicit tool-use constraints. +- Fixed Cursor streams prematurely reporting success before late CONNECT or gRPC terminal failures were observed, and resolved issues rejecting transport ends without a `turnEnded` signal. + +## [17.0.1] - 2026-07-16 + +### Fixed + +- Fixed OpenRouter cost reporting to use the provider's authoritative account charge instead of catalog token-price estimates on both Responses and Chat Completions streams. +- Fixed OpenAI Responses and Chat Completions requests forwarding unsupported sampling parameters such as `temperature` to o-series and GPT-5+ models, preventing 400 errors for mnemopi memory calls through GitHub Copilot GPT-5.6 Luna. ([#5606](https://github.com/can1357/oh-my-pi/issues/5606)) +- Fixed boolean JSON Schema subschemas (`true`/`false`) in MCP tool inputs triggering `400 INVALID_ARGUMENT` on the Google/Cloud Code Assist (Antigravity) transport by coercing them to their object equivalents (`true` → `{}`, `false` → `{ not: {} }`) before sending ([#5604](https://github.com/can1357/oh-my-pi/issues/5604)). +- Fixed thinking-enabled Claude requests routed to `google-vertex` sending the `effort-2025-11-24` beta as an `anthropic-beta` HTTP header, which Vertex rawPredict rejects with a 400. The effort beta and the `output_config.effort` field are now gated off the Vertex path the same way `context-management-2025-06-27` already is ([#5614](https://github.com/can1357/oh-my-pi/issues/5614)). +- Fixed custom and Foundry-routed Anthropic endpoints receiving first-party eager/legacy tool-streaming controls ([#5572](https://github.com/can1357/oh-my-pi/issues/5572)). +- Parsed Ollama NDJSON response bytes directly instead of decoding and buffering every network chunk as text. ([#5542](https://github.com/can1357/oh-my-pi/issues/5542)) +- Fixed Amazon Bedrock stream error handling for non-`Error` values that `JSON.stringify` cannot serialize ([#5539](https://github.com/can1357/oh-my-pi/issues/5539)). +- Fixed concurrent provider OAuth refreshes by serializing rotating-token updates across processes, fencing stale writes, and preventing background usage probes from disabling otherwise usable credentials ([#5396](https://github.com/can1357/oh-my-pi/issues/5396)). +- Fixed OpenAI Codex WebSocket connections ignoring `PI_PROXY`, provider-specific proxy settings, and standard HTTPS/ALL proxy variables ([#5384](https://github.com/can1357/oh-my-pi/issues/5384)). +- Fixed Anthropic account quota exhaustion (`This request would exceed your account's monthly spend limit`) hanging until the local deadline instead of surfacing the error: the `rate_limit_error` "spend limit" wording is now classified as a persistent usage limit, so it fails fast and rotates to a sibling credential rather than looping in the provider retry backoff. ([#4787](https://github.com/can1357/oh-my-pi/issues/4787)) +- Fixed OpenRouter daily free-model allowance errors (`free-models-per-day`) being treated as transient rate limits, so requests rotate from an exhausted API key to a healthy sibling credential. ([#4832](https://github.com/can1357/oh-my-pi/issues/4832)) + ## [17.0.0] - 2026-07-15 ### Changed diff --git a/packages/ai/package.json b/packages/ai/package.json index 019ff7821..1304533f8 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "17.0.0", + "version": "17.0.8", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/src/auth-broker/wire-schemas.ts b/packages/ai/src/auth-broker/wire-schemas.ts index f4aeacf43..d13284333 100644 --- a/packages/ai/src/auth-broker/wire-schemas.ts +++ b/packages/ai/src/auth-broker/wire-schemas.ts @@ -55,6 +55,7 @@ export const apiKeyCredentialSchema = type({ "+": "reject", type: "'api_key'", key: type("string").atLeastLength(1), + "source?": "'login'", }); /** Discriminated union accepted on POST /v1/credential (writes). */ diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index 78369ffdf..5035b1606 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -726,13 +726,19 @@ async function handleCredentialsCheck(storage: AuthStorage, signal: AbortSignal) } function handleModelsList(opts: AuthGatewayBootOptions): Response { - const list = opts.listModels ? Array.from(opts.listModels()) : []; - const data = list.map(model => ({ - id: model.id, - object: "model" as const, - owned_by: model.provider, - api: model.api, - })); + const seen = new Set(); + const data: Array<{ id: string; object: "model"; owned_by: string; api: Api }> = []; + for (const model of opts.listModels?.() ?? []) { + const id = `${model.provider}/${model.id}`; + if (seen.has(id)) continue; + seen.add(id); + data.push({ + id, + object: "model", + owned_by: model.provider, + api: model.api, + }); + } return json(200, { object: "list", data }); } diff --git a/packages/ai/src/auth-retry.ts b/packages/ai/src/auth-retry.ts index 4a140610a..b98346a82 100644 --- a/packages/ai/src/auth-retry.ts +++ b/packages/ai/src/auth-retry.ts @@ -1,6 +1,6 @@ import type { OAuthAccess } from "./auth-storage"; import * as AIError from "./error"; -import { isAuthRetryableError } from "./error/auth-classify"; +import { isAuthRetryableError, isInvalidatedOAuthTokenError } from "./error/auth-classify"; import { isUsageLimit } from "./error/flags"; import { isUsageLimitOutcome } from "./error/rate-limit"; @@ -90,7 +90,7 @@ export const AUTH_RETRY_STEPS: readonly boolean[] = [false, true]; export const AUTH_RETRY_MAX_ATTEMPTS = 64; function isDirectCredentialRotationError(error: unknown): boolean { - if (isUsageLimit(error)) return true; + if (isUsageLimit(error) || isInvalidatedOAuthTokenError(error)) return true; const status = AIError.status(error); const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; return isUsageLimitOutcome(status, message); diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index c102b5cd7..b4d8d01e7 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -11,7 +11,7 @@ import { Database, type Statement } from "bun:sqlite"; import { createHash } from "node:crypto"; import * as fs from "node:fs/promises"; import * as path from "node:path"; -import { getAgentDbPath, logger } from "@oh-my-pi/pi-utils"; +import { $env, getAgentDbPath, logger } from "@oh-my-pi/pi-utils"; import type { ApiKeyResolver } from "./auth-retry"; import * as AIError from "./error"; import { isUsageLimitOutcome } from "./error/rate-limit"; @@ -57,6 +57,8 @@ import { listCodexResetCredits, } from "./usage/openai-codex-reset"; import { opencodeGoUsageProvider } from "./usage/opencode-go"; +import { syntheticUsageProvider } from "./usage/synthetic"; +import { xaiOauthUsageProvider } from "./usage/xai-oauth"; import { zaiRankingStrategy, zaiUsageProvider } from "./usage/zai"; const USAGE_RANKING_METRIC_EPSILON = 1e-9; @@ -73,6 +75,15 @@ function fingerprintOAuthBearer(bearer: string): string { return createHash("sha256").update(bearer).digest("base64url"); } const SESSION_STICKY_CACHE_PREFIX = "session:sticky:"; +/** + * Anthropic-only idle window after which a session's pinned credential no + * longer suppresses usage-based re-ranking. Anthropic caps OAuth prompt-cache + * retention at `ttl: "1h"` (ephemeral ~5min otherwise), so after this long + * without an Anthropic resolve the conversation-prefix cache is no longer + * guaranteed warm. Other providers retain indefinite stickiness until their + * own cache lifetimes are verified. + */ +const ANTHROPIC_SESSION_STICKY_CACHE_WARM_MS = 60 * 60_000; // ───────────────────────────────────────────────────────────────────────────── // Credential Types @@ -177,7 +188,7 @@ export interface CredentialHealthResult { email?: string; /** OAuth account id if known. */ accountId?: string; - /** Organization/workspace the credential is scoped to (Anthropic multi-subscription). */ + /** Organization/workspace the credential is scoped to (Anthropic/ChatGPT multi-subscription). */ orgId?: string; orgName?: string; /** `true` when the refresh token lives on a remote broker (sentinel was present). */ @@ -587,6 +598,8 @@ const DEFAULT_USAGE_PROVIDERS: UsageProvider[] = [ opencodeGoUsageProvider, githubCopilotUsageProvider, cursorUsageProvider, + syntheticUsageProvider, + xaiOauthUsageProvider, ]; const DEFAULT_USAGE_PROVIDER_MAP = new Map( @@ -726,7 +739,7 @@ export interface OAuthAccess { projectId?: string; enterpriseUrl?: string; apiEndpoint?: string; - /** Organization/workspace the credential is scoped to (Anthropic multi-subscription). */ + /** Organization/workspace the credential is scoped to (Anthropic/ChatGPT multi-subscription). */ orgId?: string; orgName?: string; } @@ -751,7 +764,7 @@ export interface OAuthAccessFailure { projectId?: string; enterpriseUrl?: string; apiEndpoint?: string; - /** Organization/workspace the credential is scoped to (Anthropic multi-subscription). */ + /** Organization/workspace the credential is scoped to (Anthropic/ChatGPT multi-subscription). */ orgId?: string; orgName?: string; error: string; @@ -767,7 +780,7 @@ export interface OAuthAccountIdentity { accountId?: string; email?: string; projectId?: string; - /** Organization/workspace the credential is scoped to (Anthropic multi-subscription). */ + /** Organization/workspace the credential is scoped to (Anthropic/ChatGPT multi-subscription). */ orgId?: string; orgName?: string; } @@ -786,7 +799,7 @@ export interface OAuthAccountSummary { email?: string; projectId?: string; enterpriseUrl?: string; - /** Organization/workspace the credential is scoped to (Anthropic multi-subscription). */ + /** Organization/workspace the credential is scoped to (Anthropic/ChatGPT multi-subscription). */ orgId?: string; orgName?: string; } @@ -797,6 +810,8 @@ export interface InvalidateCredentialMatchingOptions { /** Options for refreshing one stored OAuth row through durable ownership. */ export interface StoredOAuthRefreshOptions { + /** Stable row id when a provider has multiple OAuth credentials. */ + credentialId?: number; observedCredential?: T; credentialFromRow: (credential: OAuthCredential) => T | undefined; forceRefresh?: boolean; @@ -1135,7 +1150,10 @@ export class AuthStorage { /** Tracks next credential index per provider:type key for round-robin distribution (non-session use). */ #providerRoundRobinIndex: Map = new Map(); /** Tracks the last used credential per provider for a session (used for rate-limit switching). */ - #sessionLastCredential: Map> = new Map(); + #sessionLastCredential: Map< + string, + Map + > = new Map(); /** Recent bearer fingerprints resolved for each durable OAuth row; used only for delayed usage-limit attribution. */ #oauthBearerFingerprints: Map> = new Map(); /** Maps provider:type -> credentialIndex -> blockedUntilMs for temporary backoff. */ @@ -1683,17 +1701,18 @@ export class AuthStorage { index: number, ): void { if (!sessionId) return; + const nowMs = Date.now(); const sessionMap = this.#sessionLastCredential.get(provider) ?? new Map(); - sessionMap.set(sessionId, { type, index }); + sessionMap.set(sessionId, { type, index, lastUsedAtMs: nowMs }); this.#sessionLastCredential.set(provider, sessionMap); try { const credentialId = this.#getStoredCredentials(provider)[index]?.id; if (credentialId !== undefined) { const cacheKey = `${SESSION_STICKY_CACHE_PREFIX}${provider}:${sessionId}`; - const cacheValue = JSON.stringify({ type, index, credentialId }); + const cacheValue = JSON.stringify({ type, index, credentialId, lastUsedAtMs: nowMs }); // Expires in 30 days - const expiresAtSec = Math.floor(Date.now() / 1000) + 30 * 24 * 60 * 60; + const expiresAtSec = Math.floor(nowMs / 1000) + 30 * 24 * 60 * 60; this.#store.setCache(cacheKey, cacheValue, expiresAtSec); } } catch (err) { @@ -1705,7 +1724,7 @@ export class AuthStorage { #getSessionCredential( provider: string, sessionId: string | undefined, - ): { type: AuthCredential["type"]; index: number } | undefined { + ): { type: AuthCredential["type"]; index: number; lastUsedAtMs?: number } | undefined { if (!sessionId) return undefined; let sessionMap = this.#sessionLastCredential.get(provider); if (sessionMap?.has(sessionId)) { @@ -1715,7 +1734,12 @@ export class AuthStorage { const cacheKey = `${SESSION_STICKY_CACHE_PREFIX}${provider}:${sessionId}`; const raw = this.#store.getCache(cacheKey); if (raw) { - const val = JSON.parse(raw) as { type: AuthCredential["type"]; index: number; credentialId?: number }; + const val = JSON.parse(raw) as { + type: AuthCredential["type"]; + index: number; + credentialId?: number; + lastUsedAtMs?: number; + }; if (val.credentialId !== undefined) { const stored = this.#getStoredCredentials(provider); @@ -1735,7 +1759,7 @@ export class AuthStorage { sessionMap = new Map(); this.#sessionLastCredential.set(provider, sessionMap); } - const sessionVal = { type: val.type, index: val.index }; + const sessionVal = { type: val.type, index: val.index, lastUsedAtMs: val.lastUsedAtMs }; sessionMap.set(sessionId, sessionVal); return sessionVal; } @@ -2007,17 +2031,32 @@ export class AuthStorage { } /** - * Persist a refreshed credential addressed by id, not a positional index. - * A concurrent disable can reorder/shrink the provider's row array while an - * async refresh is in flight, so a pre-await index is unsafe; resolving the - * row by id at write time lands the rotated token on the correct row. Returns - * the row's current index, or -1 when it was disabled/removed mid-refresh. + * Persist a refreshed credential by id only while the row still matches this + * process's snapshot. A peer rotation wins the CAS and is reloaded instead of + * being overwritten after this process releases its refresh lease. + * + * Returns the row's current index, or -1 when it was disabled or removed. */ #replaceCredentialById(provider: string, id: number, credential: AuthCredential): number { const entries = this.#getStoredCredentials(provider); const index = entries.findIndex(entry => entry.id === id); if (index === -1) return -1; - this.#store.updateAuthCredential(id, credential); + const expected = serializeCredential(provider, entries[index]!.credential); + if ( + expected && + this.#store.tryUpdateAuthCredentialIfMatches && + !this.#store.tryUpdateAuthCredentialIfMatches(id, expected.data, credential) + ) { + const latest = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + latest.map(row => ({ id: row.id, credential: row.credential })), + ); + return latest.findIndex(row => row.id === id); + } + if (!expected || !this.#store.tryUpdateAuthCredentialIfMatches) { + this.#store.updateAuthCredential(id, credential); + } const updated = [...entries]; updated[index] = { id, credential }; this.#setStoredCredentials(provider, updated); @@ -2150,7 +2189,11 @@ export class AuthStorage { provider, rows.map(row => ({ id: row.id, credential: row.credential })), ); - const row = rows.find(entry => entry.credential.type === "oauth"); + const row = rows.find( + entry => + entry.credential.type === "oauth" && + (options.credentialId === undefined || entry.id === options.credentialId), + ); if (row?.credential.type !== "oauth") { return { credential: undefined, refreshed: false, removed: false }; } @@ -2189,7 +2232,11 @@ export class AuthStorage { provider, rows.map(row => ({ id: row.id, credential: row.credential })), ); - const row = rows.find(entry => entry.credential.type === "oauth"); + const row = rows.find( + entry => + entry.credential.type === "oauth" && + (options.credentialId === undefined || entry.id === options.credentialId), + ); if (row?.credential.type !== "oauth") { return { credential: undefined, refreshed: false, removed: false }; } @@ -2266,7 +2313,7 @@ export class AuthStorage { return { credential: undefined, refreshed: false, removed: true }; } await this.reload(); - const latest = this.get(provider); + const latest = this.#getStoredCredentials(provider).find(entry => entry.id === row.id)?.credential; return { credential: latest?.type === "oauth" ? options.credentialFromRow(latest) : undefined, refreshed: false, @@ -2316,7 +2363,7 @@ export class AuthStorage { ) ) { await this.reload(); - const latest = this.get(provider); + const latest = this.#getStoredCredentials(provider).find(entry => entry.id === row.id)?.credential; return { credential: latest?.type === "oauth" ? options.credentialFromRow(latest) : undefined, refreshed: false, @@ -2808,37 +2855,27 @@ export class AuthStorage { return match?.id; } - #persistRefreshedUsageCredential(provider: Provider, previous: UsageCredential, next: UsageCredential): void { - const entries = this.#getStoredCredentials(provider); - // Same sentinel rule as #findStoredCredentialIdForUsageCredential above. - const previousRefresh = - previous.refreshToken && previous.refreshToken !== REMOTE_REFRESH_SENTINEL ? previous.refreshToken : undefined; - const index = entries.findIndex(entry => { - if (entry.credential.type !== "oauth") return false; - if (previousRefresh && entry.credential.refresh === previousRefresh) return true; - if (previous.accessToken && entry.credential.access === previous.accessToken) return true; - return ( - entry.credential.accountId === previous.accountId && - entry.credential.email === previous.email && - entry.credential.projectId === previous.projectId && - entry.credential.orgId === previous.orgId - ); - }); - if (index === -1) return; - const existing = entries[index]!.credential; - if (existing.type !== "oauth") return; - this.#replaceCredentialAt(provider, index, { + #persistRefreshedUsageCredential( + provider: Provider, + previous: UsageCredential, + next: UsageCredential, + credentialId = this.#findStoredCredentialIdForUsageCredential(provider, previous), + ): void { + if (credentialId === undefined) return; + const entry = this.#getStoredCredentials(provider).find(candidate => candidate.id === credentialId); + if (entry?.credential.type !== "oauth") return; + this.#replaceCredentialById(provider, credentialId, { type: "oauth", - access: next.accessToken ?? existing.access, - refresh: next.refreshToken ?? existing.refresh, - expires: next.expiresAt ?? existing.expires, + access: next.accessToken ?? entry.credential.access, + refresh: next.refreshToken ?? entry.credential.refresh, + expires: next.expiresAt ?? entry.credential.expires, accountId: next.accountId, projectId: next.projectId, email: next.email, enterpriseUrl: next.enterpriseUrl, apiEndpoint: next.apiEndpoint, - orgId: next.orgId ?? existing.orgId, - orgName: next.orgName ?? existing.orgName, + orgId: next.orgId ?? entry.credential.orgId, + orgName: next.orgName ?? entry.credential.orgName, }); } @@ -2878,7 +2915,12 @@ export class AuthStorage { timeoutSignal, ); const refreshedCredential = this.#mergeRefreshedUsageCredential(request.credential, refreshed); - this.#persistRefreshedUsageCredential(request.provider, request.credential, refreshedCredential); + this.#persistRefreshedUsageCredential( + request.provider, + request.credential, + refreshedCredential, + refreshableCredentialId, + ); params = { ...request, credential: refreshedCredential, @@ -2887,46 +2929,16 @@ export class AuthStorage { }; } catch (error) { const errorMsg = String(error); - // Definitive failure (invalid_grant / 401 not from a network blip) means - // the refresh token itself is dead — probing with the original credential - // will 401, the catch below will return null, and #fetchUsageCached's - // last-good fallback will surface yesterday's report indefinitely - // (including its already-elapsed `resetsAt`). CAS-disable the row and - // clear the cache so the credential drops out of the report instead of - // freezing in place until the user notices and re-logs in. - if (AIError.isDefinitiveOAuthFailure(errorMsg)) { - const credentialId = this.#findStoredCredentialIdForUsageCredential( - request.provider, - request.credential, - ); - if (credentialId !== undefined) { - const entries = this.#getStoredCredentials(request.provider); - const index = entries.findIndex(entry => entry.id === credentialId); - if (index !== -1) { - const disabled = this.#tryDisableCredentialAtIfMatches( - request.provider, - index, - refreshableCredential, - `oauth refresh failed during usage probe: ${errorMsg}`, - ); - if (disabled) { - this.#usageLogger?.warn( - "Usage credential refresh failed definitively; credential disabled", - { provider: request.provider, credentialId, error: errorMsg }, - ); - // Neutralize last-good for this cache key: write a null - // entry with an immediately-elapsed expiry so a future - // getStale lookup (e.g. on re-login under the same - // account identity) can't replay the stale report. - this.#usageCache.set(this.#buildUsageReportCacheKey(request), { - value: null, - expiresAt: 0, - }); - return null; - } - } - } + if (request.credential.expiresAt <= Date.now() && AIError.isDefinitiveOAuthFailure(errorMsg)) { + // The current access token is unusable, so don't replay an + // old usage report after its rotating refresh token is revoked. + // This changes cache state only; usage polling remains + // non-authoritative about the credential lifecycle. + this.#usageCache.set(this.#buildUsageReportCacheKey(request), { value: null, expiresAt: 0 }); } + // Usage polling is advisory. A refresh can fail while the current + // access token remains valid inside the refresh skew, so probe with + // that token and never mutate credential state from this path. this.#usageLogger?.debug("Usage credential refresh failed, using original credential", { provider: request.provider, error: errorMsg, @@ -3219,6 +3231,29 @@ export class AuthStorage { entries = dedupedEntries; } + // SuperGrok billing only accepts OAuth bearers. Catalog envVars for + // xai-oauth are [XAI_OAUTH_TOKEN, XAI_API_KEY], so the generic path + // would (a) build api_key usage requests from stored keys / the paid + // API env var and (b) never fall through to XAI_OAUTH_TOKEN when a + // non-OAuth row is the only stored credential. Skip api_key material + // and only env-fallback to the dedicated OAuth bearer. + if (providerId === "xai-oauth") { + let hasUsableStoredOAuthCredential = false; + for (const entry of entries) { + if (entry.credential.type !== "oauth") continue; + const request = this.#buildUsageRequestForOauth(provider, entry.credential, baseUrl); + if (providerImpl.supports && !providerImpl.supports(request)) continue; + requests.push(request); + hasUsableStoredOAuthCredential = true; + } + const oauthToken = $env.XAI_OAUTH_TOKEN?.trim(); + if (!hasUsableStoredOAuthCredential && oauthToken) { + const request = this.#buildUsageRequest(provider, { type: "oauth", accessToken: oauthToken }, baseUrl); + if (!providerImpl.supports || providerImpl.supports(request)) requests.push(request); + } + continue; + } + if (entries.length === 0) { const runtimeKey = this.#runtimeOverrides.get(providerId); const envKey = getEnvApiKey(providerId); @@ -3275,15 +3310,14 @@ export class AuthStorage { const identifiers: string[] = []; const email = this.#getUsageReportMetadataValue(report, "email"); if (email) identifiers.push(`email:${email.toLowerCase()}`); - if (report.provider === "anthropic") { - // Anthropic: one account email can hold several organizations - // (Team seat + personal Max). Reports from different orgs must not - // merge — scope every identifier by org when the report carries one. - // When the email could not be recovered, fall back to the account - // (identical across orgs, hence the org qualifier is what keeps two - // subscriptions apart) so no-email reports still merge per org. - // Org-less reports (pre-upgrade caches) keep their bare identifiers - // and only merge among themselves. + if (report.provider === "anthropic" || report.provider === "openai-codex") { + // One account email can hold several org-scoped subscriptions + // (Anthropic organizations, ChatGPT workspaces). Reports from + // different orgs must not merge — scope every identifier by org + // when the report carries one; fall back to the account when the + // email could not be recovered so no-email reports still merge + // per org. Org-less reports (pre-upgrade caches) keep their bare + // identifiers and only merge among themselves. if (identifiers.length === 0) { const accountId = this.#getUsageReportMetadataValue(report, "accountId") ?? this.#getUsageReportScopeAccountId(report); @@ -3291,12 +3325,11 @@ export class AuthStorage { } const orgId = this.#getUsageReportMetadataValue(report, "orgId"); if (orgId) { - if (identifiers.length === 0) return [`anthropic:org:${orgId.toLowerCase()}`]; - return identifiers.map(identifier => `anthropic:org:${orgId.toLowerCase()}|${identifier.toLowerCase()}`); + if (identifiers.length === 0) return [`${report.provider}:org:${orgId.toLowerCase()}`]; + return identifiers.map( + identifier => `${report.provider}:org:${orgId.toLowerCase()}|${identifier.toLowerCase()}`, + ); } - return identifiers.map(identifier => `anthropic:${identifier.toLowerCase()}`); - } - if (report.provider === "openai-codex") { return identifiers.map(identifier => `${report.provider}:${identifier.toLowerCase()}`); } const projectId = @@ -3667,6 +3700,7 @@ export class AuthStorage { row.provider as Provider, initialRequest.credential, refreshedCredential, + row.id, ); params = { ...params, @@ -3895,16 +3929,15 @@ export class AuthStorage { * how fast the window's remaining quota must be consumed to fully use it * before it resets and expires. Higher = more headroom at risk of expiring * unused = ranked first, so selection chases quota that is about to be - * wasted ("use it or lose it"). Without a reset clock the headroom - * fraction alone is returned, degrading to most-headroom-first. + * wasted ("use it or lose it"). Without a reset clock, the full window + * duration is assumed to remain so clocked and clockless scores stay comparable. */ #computeWindowRequiredDrain(limit: UsageLimit | undefined, nowMs: number, fallbackDurationMs: number): number { const headroom = 1 - this.#normalizeUsageFraction(limit); if (headroom <= 0) return 0; const resetAt = this.#resolveWindowResetAt(limit?.window); - if (resetAt === undefined) return headroom; const durationMs = limit?.window?.durationMs ?? fallbackDurationMs; - let remainingMs = resetAt - nowMs; + let remainingMs = resetAt === undefined ? durationMs : resetAt - nowMs; if (Number.isFinite(durationMs) && durationMs > 0) { remainingMs = Math.min(remainingMs, durationMs); } @@ -4144,16 +4177,39 @@ export class AuthStorage { sessionPreferredCredential !== undefined && (sessionPreferredCredential.refresh.trim().length > 0 || Date.now() + OAUTH_REFRESH_SKEW_MS < sessionPreferredCredential.expires); - // Skip ranking only when the session already has a working preferred credential — re-ranking - // mid-session causes account switches that cold-start the server-side prompt cache. New sessions - // (no preference) and sessions whose preferred is blocked still rank, so we pick the account - // with the most headroom proactively and fall back intelligently when rate-limited. + // Skip ranking when the session already has a working preferred credential and its prompt + // cache may still be warm. Only Anthropic has a verified idle boundary here; unverified + // providers retain indefinite stickiness rather than risk switching while their prompt cache + // remains warm. New Anthropic sessions (no preference), sessions whose preferred is blocked, + // and sessions idle past {@link ANTHROPIC_SESSION_STICKY_CACHE_WARM_MS} still rank. Legacy + // pins predating `lastUsedAtMs` count as warm until the next resolve rewrites the row. + const sessionPreferredLastUsedAtMs = + sessionCredential?.type === "oauth" ? sessionCredential.lastUsedAtMs : undefined; + const sessionPreferredIsWarm = + provider !== "anthropic" || + sessionPreferredLastUsedAtMs === undefined || + Date.now() - sessionPreferredLastUsedAtMs < ANTHROPIC_SESSION_STICKY_CACHE_WARM_MS; const sessionPreferredIsAvailable = sessionPreferredIndex !== undefined && sessionPreferredCanRefreshOrUse && !this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope); - const shouldRank = checkUsage && (!sessionPreferredIsAvailable || hasPlanRequirement); - const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; + const shouldRank = checkUsage && (!sessionPreferredIsAvailable || !sessionPreferredIsWarm || hasPlanRequirement); + // When ranking, seed the pinned credential first in the evaluation order so it wins genuine + // ties (the ranked comparator falls back to `orderPos`) without overriding a strictly-better + // sibling — this respects the residual value of a same-account shared static prefix that other + // workspace traffic may have kept warm, while still rotating away from a clearly-worse account. + const baseRankingOrder = credentials.map((_credential, index) => index); + let rankingOrder = shouldRank && sessionId ? baseRankingOrder : order; + const sessionPreferredRankingPos = + shouldRank && sessionId && sessionPreferredIndex !== undefined && !hasPlanRequirement + ? credentials.findIndex(entry => entry.index === sessionPreferredIndex) + : -1; + if (sessionPreferredRankingPos > 0) { + rankingOrder = [ + sessionPreferredRankingPos, + ...baseRankingOrder.filter(index => index !== sessionPreferredRankingPos), + ]; + } const candidates = shouldRank ? await this.#rankOAuthSelections({ providerKey, @@ -4171,7 +4227,10 @@ export class AuthStorage { .filter((selection): selection is { credential: OAuthCredential; index: number } => Boolean(selection)) .map(selection => ({ selection, usage: null, usageChecked: false })); - if (sessionPreferredIndex !== undefined && !hasPlanRequirement) { + // On the warm skip path the candidate list follows the round-robin `order`, not the pin, so + // hoist the pinned credential to the front to actually reuse it. When ranking ran, the pin is + // already a mere tie-break via `rankingOrder`; do not override the ranked result here. + if (!shouldRank && sessionPreferredIndex !== undefined && !hasPlanRequirement) { const sessionPreferredCandidate = candidates.findIndex( candidate => !this.#isCredentialBlocked(provider, providerKey, candidate.selection.index, blockScope) && @@ -4258,6 +4317,29 @@ export class AuthStorage { hasPlanRequirement && candidates.some(candidate => getOpenAICodexPlanEligibility(candidate.usage, planRequirement) === true); + // Plan-gated Codex models rank on every resolve to re-verify account tiers, + // so the drain-urgency order can flip between two eligible accounts as their + // usage headroom shifts. Promote the session-preferred credential back to the + // front while it is unblocked and still plan-eligible (or the requirement is + // unenforced and the pin is not known-ineligible) so an active session never + // silently migrates accounts mid-conversation; blocked, exhausted, or + // known-ineligible pins still fall through to the ranked sibling. + if (hasPlanRequirement && sessionPreferredIndex !== undefined) { + const sessionPreferredCandidate = candidates.findIndex( + candidate => + !this.#isCredentialBlocked(provider, providerKey, candidate.selection.index, blockScope) && + candidate.selection.index === sessionPreferredIndex, + ); + if (sessionPreferredCandidate > 0) { + const preferred = candidates[sessionPreferredCandidate]!; + const planEligibility = getOpenAICodexPlanEligibility(preferred.usage, planRequirement); + if (planEligibility === true || (!enforcePlanRequirement && planEligibility !== false)) { + candidates.splice(sessionPreferredCandidate, 1); + candidates.unshift(preferred); + } + } + } + const passes: Array<{ allowBlocked: boolean; enforcePlanRequirement: boolean }> = [ { allowBlocked: false, enforcePlanRequirement }, { allowBlocked: true, enforcePlanRequirement }, @@ -4317,6 +4399,42 @@ export class AuthStorage { credential: OAuthCredential, credentialId: number | undefined, signal?: AbortSignal, + ): Promise { + const hasDurableLease = + !!this.#store.tryAcquireCredentialRefreshLease && + !!this.#store.getCredentialRefreshLeaseExpiresAt && + !!this.#store.releaseCredentialRefreshLease && + !!this.#store.renewCredentialRefreshLease; + if (credentialId !== undefined && hasDurableLease) { + const forceRefresh = credential.expires === 0; + const result = await this.refreshStoredOAuthCredential(provider, { + credentialId, + observedCredential: forceRefresh ? undefined : credential, + credentialFromRow: row => row, + forceRefresh, + signal, + refresh: (current, refreshSignal) => + this.#requestOAuthCredentialRefresh( + provider, + current, + credentialId, + signal && refreshSignal ? AbortSignal.any([signal, refreshSignal]) : (signal ?? refreshSignal), + ), + }); + if (result.credential) return result.credential; + throw new AIError.OAuthError(`OAuth credential no longer exists for provider: ${provider}`, { + kind: "token-refresh", + provider, + }); + } + return this.#requestOAuthCredentialRefresh(provider, credential, credentialId, signal); + } + + async #requestOAuthCredentialRefresh( + provider: Provider, + credential: OAuthCredential, + credentialId: number | undefined, + signal?: AbortSignal, ): Promise { let refreshPromise: Promise; // Caller override > store-level hook > local per-provider refresh. @@ -4343,10 +4461,9 @@ export class AuthStorage { // Bound the refresh so a slow/hanging token endpoint cannot stall credential selection. // Caller-driven abort jumps the gun on the timeout — the agent's ESC must // take priority over the floor timeout. - let timeout: NodeJS.Timeout | undefined; - let onAbort: (() => void) | undefined; const cancellation = Promise.withResolvers(); - timeout = setTimeout( + let onAbort: (() => void) | undefined; + const timeout = setTimeout( () => cancellation.reject( new AIError.OAuthError(`OAuth token refresh timed out for provider: ${provider}`, { @@ -4367,7 +4484,7 @@ export class AuthStorage { try { return await Promise.race([refreshPromise, cancellation.promise]); } finally { - if (timeout) clearTimeout(timeout); + clearTimeout(timeout); if (signal && onAbort) signal.removeEventListener("abort", onAbort); } } @@ -5202,9 +5319,15 @@ export class AuthStorage { if (credential.type !== "oauth") continue; const credentialEmail = credential.email?.trim().toLowerCase(); const credentialAccountId = credential.accountId?.trim().toLowerCase(); - if ((email && credentialEmail === email) || (accountId && credentialAccountId === accountId)) { - matches.push(entry.id); - } + // Every identity dimension present on BOTH sides must agree — the + // account id is shared workspace-wide and one email can span + // workspaces, so a single-dimension match can cross-link siblings. + const emailComparable = Boolean(email && credentialEmail); + const accountComparable = Boolean(accountId && credentialAccountId); + if (!emailComparable && !accountComparable) continue; + if (emailComparable && credentialEmail !== email) continue; + if (accountComparable && credentialAccountId !== accountId) continue; + matches.push(entry.id); } return matches; } @@ -5359,6 +5482,21 @@ export class AuthStorage { Date.now() + AuthStorage.#defaultBackoffMs, ); + if (target && AIError.isInvalidatedOAuthTokenError(error)) { + const disabledCause = message ?? "upstream reported invalidated OAuth token"; + const deleted = this.#store.deleteAuthCredentialRemote + ? await this.#store.deleteAuthCredentialRemote(target.id, disabledCause) + : this.disableCredentialById(target.id, disabledCause); + if (deleted) { + const latestRows = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + latestRows.map(row => ({ id: row.id, credential: row.credential })), + ); + } + return deleted && hasSibling; + } + if (target) { const markSuspect = this.#store.markCredentialSuspect?.bind(this.#store); if (markSuspect) { @@ -5799,16 +5937,15 @@ function toStoredAuthCredential(row: AuthRow, credential: AuthCredential): Store function resolveProviderCredentialIdentityKey(provider: string, identifiers: string[]): string | null { const emailIdentifier = identifiers.find(identifier => identifier.startsWith("email:")); - if (provider === "anthropic") { - // One Anthropic account email can hold several organizations (e.g. a - // Team seat plus a personal Max plan), each with its own org-scoped - // token and limit pools. Scope identity by org so both subscriptions - // can be stored side by side. The qualifier rides on whichever base - // identity is available — the account UUID is IDENTICAL across the - // orgs of one login account, so an unqualified account/project - // fallback would still collapse two subscriptions whenever the email - // could not be recovered. Org-less credentials (rows written before - // org capture existed) keep their bare key. + if (provider === "anthropic" || provider === "openai-codex") { + // One account email can hold several organizations/workspaces (e.g. a + // Team seat plus a personal plan), each with its own org-scoped token + // and limit pools. Scope identity by org so both subscriptions can be + // stored side by side. The qualifier rides on whichever base identity + // is available, so an unqualified account/project fallback would + // still collapse two subscriptions whenever the email could not be + // recovered. Org-less credentials (rows written before org capture + // existed) keep their bare key. const base = emailIdentifier ?? identifiers.find(identifier => identifier.startsWith("account:")) ?? @@ -5818,7 +5955,6 @@ function resolveProviderCredentialIdentityKey(provider: string, identifiers: str // No base identity at all: the org alone still distinguishes the row. return orgIdentifier ?? null; } - if (provider === "openai-codex" && emailIdentifier) return emailIdentifier; const accountIdentifier = identifiers.find(identifier => identifier.startsWith("account:")); if (accountIdentifier) return accountIdentifier; if (emailIdentifier) return emailIdentifier; @@ -5855,9 +5991,9 @@ function matchesReplacementCredential( if (incomingIdentityKey === existingIdentityKey) return true; if (existingIdentityKey === null) return false; // One-way upgrade, applied only when the INCOMING identity key carries the - // org qualifier (only anthropic keys do, so other providers never reach the - // checks below). An org-scoped login `org:` claims (and re-keys) any - // existing row that denotes the same subscription: + // org qualifier (only anthropic and openai-codex keys do, so other + // providers never reach the checks below). An org-scoped login `org:` + // claims (and re-keys) any existing row that denotes the same subscription: // - `org:` — org-only row stored when identity recovery failed, claimed // once a later same-org login recovers a base identity; // - `` for any base identity `` (email/account/project) the incoming @@ -5881,10 +6017,16 @@ function matchesReplacementCredential( existing.type === "oauth" && existingIdentityKey.endsWith(`|${orgIdentifier}`) ? extractOAuthCredentialIdentifiers(existing) : null; + // A base identifier that merely repeats the org qualifier's id carries no + // per-user identity (openai-codex stores the ChatGPT workspace id as both + // accountId and orgId, shared by every member) — letting it act as a + // claimable base would re-key another member's same-org row. + const orgQualifierId = orgIdentifier.slice("org:".length); for (const identifier of incomingIdentifiers) { const isBase = identifier.startsWith("email:") || identifier.startsWith("account:") || identifier.startsWith("project:"); if (!isBase) continue; + if (identifier.slice(identifier.indexOf(":") + 1) === orgQualifierId) continue; if (existingIdentityKey === identifier) return true; if (existingIdentityKey === `${identifier}|${orgIdentifier}`) return true; if (existingIdentifiers?.includes(identifier)) return true; diff --git a/packages/ai/src/dialect/owned-stream.ts b/packages/ai/src/dialect/owned-stream.ts index e5990b926..2c40aa961 100644 --- a/packages/ai/src/dialect/owned-stream.ts +++ b/packages/ai/src/dialect/owned-stream.ts @@ -109,6 +109,9 @@ export function wrapInbandToolStream( case "thinking_end": projector?.thinkingEnd(); break; + case "image_end": + projector?.keep(event.content); + break; case "text_delta": // `text()` returns true once the model starts fabricating its own // tool result. In abort mode we cut the turn immediately so the @@ -206,6 +209,14 @@ class InbandStreamProjector { this.#closeText(); this.#closeThinking(); this.#partial.content.push(block); + if (this.#emitEvents && block.type === "image") { + this.#out.push({ + type: "image_end", + contentIndex: this.#partial.content.length - 1, + content: block, + partial: this.#partial, + }); + } } // Forward a native tool call's lifecycle live. `source` comes from the inner diff --git a/packages/ai/src/dialect/thinking.ts b/packages/ai/src/dialect/thinking.ts index 973b5b136..fcfd222c1 100644 --- a/packages/ai/src/dialect/thinking.ts +++ b/packages/ai/src/dialect/thinking.ts @@ -32,6 +32,16 @@ export class ThinkingInbandScanner implements InbandScanner { #thinking = ""; /** Fence-aware close-matcher while inside a ` ```thinking ` block; undefined otherwise. */ #fenced: FencedThinkingScanner | undefined; + /** Backtick count that opened the Markdown code span/fence we are inside; 0 when not in code. */ + #codeTicks = 0; + /** True when {@link #codeTicks} opened a fenced block (closes on a fence line), not an inline span. */ + #codeFenced = false; + /** + * Leading-space count on the current output line, or -1 once a non-space + * character has appeared. Starts at 0 (line start) so a fence opening the + * stream — or one indented ≤3 spaces, as CommonMark allows — is recognized. + */ + #lineIndent = 0; feed(text: string): InbandScanEvent[] { if (text.length === 0) return []; @@ -86,25 +96,93 @@ export class ThinkingInbandScanner implements InbandScanner { this.#closeTag = ""; continue; } - - const tag = findEarliestOpen(this.#buffer); - if (!tag) { - const hold = final ? 0 : partialSuffixOverlapAny(this.#buffer, OPENS); - const emit = this.#buffer.slice(0, this.#buffer.length - hold); - if (emit.length > 0) events.push({ type: "text", text: emit }); - this.#buffer = this.#buffer.slice(this.#buffer.length - hold); + if (this.#codeTicks > 0) { + if (this.#emitCode(final, events)) continue; break; } - if (tag.index > 0) events.push({ type: "text", text: this.#buffer.slice(0, tag.index) }); - this.#buffer = this.#buffer.slice(tag.index + tag.open.length); - this.#closeTag = tag.close; + + const hit = scanVisible(this.#buffer, final); + if (hit.kind === "none") { + this.#emitText(this.#buffer, events); + this.#buffer = ""; + break; + } + if (hit.index > 0) this.#emitText(this.#buffer.slice(0, hit.index), events); + if (hit.kind === "hold") { + this.#buffer = this.#buffer.slice(hit.index); + break; + } + if (hit.kind === "code") { + const fenced = hit.ticks >= 3 && this.#lineIndent >= 0 && this.#lineIndent <= 3; + this.#emitText(this.#buffer.slice(hit.index, hit.index + hit.ticks), events); + this.#buffer = this.#buffer.slice(hit.index + hit.ticks); + this.#codeTicks = hit.ticks; + this.#codeFenced = fenced; + continue; + } + this.#buffer = this.#buffer.slice(hit.index + hit.tag.open.length); + this.#closeTag = hit.tag.close; this.#thinking = ""; - if (tag.fenced) this.#fenced = new FencedThinkingScanner(); + if (hit.tag.fenced) this.#fenced = new FencedThinkingScanner(); events.push({ type: "thinkingStart" }); } return events; } + /** + * Emit buffered content while inside a Markdown code region, suppressing + * reasoning-tag detection. A fenced block closes only on a fence line (a line + * of backticks ≥ the opener); an inline span closes on the first backtick run + * of exactly the opener length. Returns true when the region closed and the + * loop should continue, false when it held back and should break. + */ + #emitCode(final: boolean, events: InbandScanEvent[]): boolean { + if (this.#codeFenced) { + const end = findFenceCloseEnd(this.#buffer, this.#codeTicks, final); + if (end !== -1) { + this.#emitText(this.#buffer.slice(0, end), events); + this.#buffer = this.#buffer.slice(end); + this.#codeTicks = 0; + this.#codeFenced = false; + return true; + } + if (final) { + this.#emitText(this.#buffer, events); + this.#buffer = ""; + this.#codeTicks = 0; + this.#codeFenced = false; + return false; + } + // Stream committed lines; hold only the last (possibly partial) fence line. + const lastNl = this.#buffer.lastIndexOf("\n"); + if (lastNl !== -1) { + this.#emitText(this.#buffer.slice(0, lastNl + 1), events); + this.#buffer = this.#buffer.slice(lastNl + 1); + } + return false; + } + const close = findBacktickRun(this.#buffer, 0, this.#codeTicks); + if (close !== -1 && (final || close + this.#codeTicks < this.#buffer.length)) { + this.#emitText(this.#buffer.slice(0, close + this.#codeTicks), events); + this.#buffer = this.#buffer.slice(close + this.#codeTicks); + this.#codeTicks = 0; + return true; + } + // No committed close yet: emit text, holding a trailing backtick run that + // may still grow into — or past — the closing delimiter. + const hold = final ? 0 : trailingBacktickRun(this.#buffer); + this.#emitText(this.#buffer.slice(0, this.#buffer.length - hold), events); + this.#buffer = this.#buffer.slice(this.#buffer.length - hold); + if (final) this.#codeTicks = 0; + return false; + } + + #emitText(text: string, events: InbandScanEvent[]): void { + if (text.length === 0) return; + events.push({ type: "text", text }); + this.#lineIndent = trailingLineIndent(text, this.#lineIndent); + } + #emitThinking(delta: string, events: InbandScanEvent[]): void { if (delta.length === 0) return; this.#thinking += delta; @@ -112,11 +190,103 @@ export class ThinkingInbandScanner implements InbandScanner { } } -function findEarliestOpen(buffer: string): (Tag & { index: number }) | undefined { - let best: (Tag & { index: number }) | undefined; - for (const tag of TAGS) { - const index = buffer.indexOf(tag.open); - if (index !== -1 && (!best || index < best.index)) best = { ...tag, index }; +/** Outcome of scanning idle visible text for the next reasoning-tag or code-span boundary. */ +type VisibleHit = + | { readonly kind: "tag"; readonly index: number; readonly tag: Tag } + | { readonly kind: "code"; readonly index: number; readonly ticks: number } + | { readonly kind: "hold"; readonly index: number } + | { readonly kind: "none" }; + +/** + * Walk idle visible text for the earliest boundary: a leaked reasoning-tag open, + * a Markdown code-span/fence opener (a backtick run), or — when more chunks may + * follow — a held partial delimiter at the buffer tail. + * + * Reasoning tags win at any position so the gemini ` ```thinking ` fence is + * healed instead of being read as a code fence. Backtick runs enter code mode so + * a literal `` inside inline code or a fenced block stays visible text. + */ +function scanVisible(buffer: string, final: boolean): VisibleHit { + for (let i = 0; i < buffer.length; i++) { + const tag = TAGS.find(candidate => buffer.startsWith(candidate.open, i)); + if (tag) return { kind: "tag", index: i, tag }; + if (!final) { + const rest = buffer.slice(i); + if (OPENS.some(open => open.length > rest.length && open.startsWith(rest))) { + return { kind: "hold", index: i }; + } + } + if (buffer[i] === "`") { + const ticks = backtickRun(buffer, i); + if (!final && i + ticks === buffer.length) return { kind: "hold", index: i }; + return { kind: "code", index: i, ticks }; + } } - return best; + return { kind: "none" }; +} + +/** Length of the maximal backtick run beginning at `from`. */ +function backtickRun(buffer: string, from: number): number { + let end = from; + while (end < buffer.length && buffer[end] === "`") end++; + return end - from; +} + +/** Index of the first maximal backtick run of exactly `ticks` at/after `from`, else -1. */ +function findBacktickRun(buffer: string, from: number, ticks: number): number { + for (let i = buffer.indexOf("`", from); i !== -1; i = buffer.indexOf("`", i)) { + const run = backtickRun(buffer, i); + if (run === ticks) return i; + i += run; + } + return -1; +} + +/** Length of a backtick run that ends at the buffer tail; 0 when the tail is not a backtick. */ +function trailingBacktickRun(buffer: string): number { + let start = buffer.length; + while (start > 0 && buffer[start - 1] === "`") start--; + return buffer.length - start; +} + +/** + * Leading-space count of the line at the tail of `text`, continuing from the + * prior line's `indent` state (see {@link ThinkingInbandScanner.#lineIndent}). + * Returns -1 once any non-space character has appeared on the current line. + */ +function trailingLineIndent(text: string, prior: number): number { + const lastNl = text.lastIndexOf("\n"); + let indent = lastNl === -1 ? prior : 0; + for (let i = lastNl + 1; i < text.length; i++) { + if (indent === -1) break; + indent = text[i] === " " ? indent + 1 : -1; + } + return indent; +} + +/** + * Index just past the first closing fence line for a fenced block opened with + * `ticks` backticks, or -1 when none is committed yet. A closing fence is a whole + * line whose trimmed content is only backticks, at least `ticks` of them. A line + * without a terminating newline is committed only when `final` (no more input can + * extend it into a non-fence line). + */ +function findFenceCloseEnd(buffer: string, ticks: number, final: boolean): number { + for (let start = 0; start <= buffer.length; ) { + const nl = buffer.indexOf("\n", start); + const terminated = nl !== -1; + const line = buffer.slice(start, terminated ? nl : buffer.length).trim(); + if (line.length >= ticks && isAllBackticks(line) && (terminated || final)) { + return terminated ? nl + 1 : buffer.length; + } + if (!terminated) break; + start = nl + 1; + } + return -1; +} + +/** True when `text` is non-empty and every character is a backtick. */ +function isAllBackticks(text: string): boolean { + for (let i = 0; i < text.length; i++) if (text[i] !== "`") return false; + return text.length > 0; } diff --git a/packages/ai/src/error/auth-classify.ts b/packages/ai/src/error/auth-classify.ts index 9234ada9b..e23db5c81 100644 --- a/packages/ai/src/error/auth-classify.ts +++ b/packages/ai/src/error/auth-classify.ts @@ -12,6 +12,18 @@ export function isDefinitiveOAuthFailure(errorMsg: string): boolean { return isOAuthExpiry(errorMsg); } +const INVALIDATED_OAUTH_TOKEN_PATTERN = /\binvalidated oauth token\b/i; + +/** Whether an upstream response explicitly says the supplied OAuth bearer was invalidated. */ +export function isInvalidatedOAuthTokenError(error: unknown): boolean { + if (typeof error === "object" && error !== null && "errorMessage" in error) { + const errorMessage = error.errorMessage; + if (typeof errorMessage === "string" && INVALIDATED_OAUTH_TOKEN_PATTERN.test(errorMessage)) return true; + } + const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; + return message !== undefined && INVALIDATED_OAUTH_TOKEN_PATTERN.test(message); +} + /** * Whether an upstream failure should rotate to a sibling credential: a hard * `401`, a body-classified usage limit (Codex `usage_limit_reached`, Anthropic @@ -22,6 +34,7 @@ export function isDefinitiveOAuthFailure(errorMsg: string): boolean { */ export function isAuthRetryableError(error: unknown): boolean { if (isUsageLimit(error)) return true; + if (isInvalidatedOAuthTokenError(error)) return true; const httpStatus = extractHttpStatusFromError(error); if (httpStatus === 401) return true; const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; diff --git a/packages/ai/src/error/flags.ts b/packages/ai/src/error/flags.ts index 811713b60..a67f56a12 100644 --- a/packages/ai/src/error/flags.ts +++ b/packages/ai/src/error/flags.ts @@ -7,7 +7,7 @@ import { ProviderHttpError, STREAM_ENVELOPE_ERROR_PREFIX, } from "./classes"; -import { isOpaqueStatusBody, matchesUsageLimitText, parseRateLimitReason } from "./rate-limit"; +import { isOpaqueStatusBody, isUsageLimitStatus, matchesUsageLimitText, parseRateLimitReason } from "./rate-limit"; export const Flag = { Class: 0x1000, @@ -90,7 +90,7 @@ const TRANSIENT_ENVELOPE_PATTERN = /anthropic stream envelope error:/i; const TRANSIENT_ENVELOPE_BEFORE_START_PATTERN = /before message_start/i; export const STREAM_READ_ERROR_PATTERN = /stream[_ -]?read[_ -]?error/i; export const TRANSIENT_TRANSPORT_PATTERN = - /overloaded|provider.?returned.?error|rate.?limit|too many requests|429|500|502|503|504|service.?unavailable|server.?error|internal.?error|retry your request|network.?error|connection.?error|connection.?refused|other side closed|fetch failed|upstream.?connect|upstream.?request.?failed|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall|no error details in response|HTTP2(?:StreamReset|RefusedStream|EnhanceYourCalm)|malformed.?function.?call/i; + /overloaded|provider.?returned.?error|rate.?limit|too many requests|429|500|502|503|504|service.?unavailable|server.?error|internal.?error|retry your request|network.?error|connection.?error|connection.?refused|unable.?to.?connect\.\s*is the computer able to access the url\?|other side closed|fetch failed|upstream.?connect|upstream.?request.?failed|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall|no error details in response|HTTP2(?:StreamReset|RefusedStream|EnhanceYourCalm)|malformed.?function.?call/i; const AUTH_FAILURE_PATTERN = /\b(?:401|403|unauthorized|forbidden|authentication|auth[_ ]?unavailable|no auth available|(?:invalid|no)[_ ]?api[_ ]?key)\b/i; const MALFORMED_FUNCTION_CALL_PATTERN = /\bmalformed.?function.?call\b/i; @@ -318,7 +318,7 @@ function classifyText(errorMessage: string | undefined, errorStatus: number | un const cleanMessage = errorMessage; const isOpaque = isOpaqueStatusBody(cleanMessage); - const isLimitStatus = statusClean === 429; + const isLimitStatus = isUsageLimitStatus(statusClean); if ( matchesUsageLimitText(cleanMessage) || (isLimitStatus && (isOpaque || parseRateLimitReason(cleanMessage) === "QUOTA_EXHAUSTED")) @@ -505,10 +505,19 @@ export function stringify(id: number | undefined): string { const STREAM_PARSE_TRUNCATION_PATTERN = /unterminated string|unexpected end of json input|unexpected end of data|unexpected eof|end of file|eof while parsing|truncated/i; +const STREAM_PARSE_DIAGNOSTIC_PATTERN = + /(?:json parse error:\s*(?:unterminated string|unexpected end of json input|unexpected end of data|unexpected eof|end of file|eof while parsing|truncated)|json\.parse:\s*(?:unterminated string|unexpected end of data)|unexpected end of json input|unexpected eof|eof while parsing)/i; const STREAM_EVENT_ORDER_PATTERN = /stream event order|before message_start/i; -/** Transient stream corruption where the response was truncated mid-JSON. */ +/** + * Transient stream corruption where the response was truncated mid-JSON. + * + * Strings (persisted `stopDetails.explanation`/`errorMessage` diagnostics) are matched with the + * stricter {@link STREAM_PARSE_DIAGNOSTIC_PATTERN} — bare "truncated"/"end of file" text is too + * low-signal to trust once detached from a live transport `Error`, which keeps the broad pattern. + */ export function isTransientStreamParseError(error: unknown): boolean { + if (typeof error === "string") return STREAM_PARSE_DIAGNOSTIC_PATTERN.test(error); return error instanceof Error && STREAM_PARSE_TRUNCATION_PATTERN.test(error.message); } diff --git a/packages/ai/src/error/rate-limit.ts b/packages/ai/src/error/rate-limit.ts index cef4dca02..48c03ba43 100644 --- a/packages/ai/src/error/rate-limit.ts +++ b/packages/ai/src/error/rate-limit.ts @@ -19,6 +19,8 @@ const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s const ACCOUNT_RATE_LIMIT_PATTERN = /\baccount(?:'s)?\b[^\n]{0,80}\brate.?limit\b|\brate.?limit\b[^\n]{0,80}\baccount\b/i; const INSUFFICIENT_BALANCE_PATTERN = /insufficient.?balance/i; +const SPEND_LIMIT_PATTERN = /spend.?limit/i; +const OPENROUTER_DAILY_FREE_LIMIT_PATTERN = /\bfree[-_ ]models[-_ ]per[-_ ]day\b/i; /** * Classify a rate-limit error message into a reason category. @@ -54,6 +56,14 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason { return "QUOTA_EXHAUSTED"; } + if (SPEND_LIMIT_PATTERN.test(errorMessage)) { + return "QUOTA_EXHAUSTED"; + } + + if (OPENROUTER_DAILY_FREE_LIMIT_PATTERN.test(errorMessage)) { + return "QUOTA_EXHAUSTED"; + } + if ( lower.includes("per minute") || lower.includes("rate limit") || @@ -106,16 +116,19 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number { /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */ const USAGE_LIMIT_PATTERN = - /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?(?:exceeded|reached|insufficient)|额度不足|额度耗尽|resource.?exhausted|exhausted your capacity|quota will reset|insufficient.?(?:balance|quota)|run out of credits|out of credits|spending[- _]?limit|personal-team-blocked/i; + /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?(?:exceeded|reached|insufficient)|额度不足|额度耗尽|resource.?exhausted|exhausted your capacity|quota will reset|insufficient.?(?:balance|quota)|balance.?exhausted|run out of credits|out of credits|spending[- _]?limit|personal-team-blocked/i; /** * HTTP status codes that, absent richer body classification, represent an * account-local usage cap rather than a bad credential or a transient blip. + * HTTP 402 Payment Required is categorically an account-billing cap (xAI + * Grok Build "usage balance exhausted", DeepSeek "Insufficient Balance", + * OpenRouter credit exhaustion) — never a transient blip or bad credential. * Always combine with {@link isUsageLimitOutcome} when a message is available * — a 429 carrying transient rate-limit wording is NOT a usage cap. */ export function isUsageLimitStatus(status: number | undefined): boolean { - return status === 429; + return status === 429 || status === 402; } /** @@ -125,7 +138,7 @@ export function isUsageLimitStatus(status: number | undefined): boolean { * 1. Body matches {@link isUsageLimitError} (Codex `usage_limit_reached`, * Anthropic account rate-limit, Google `resource_exhausted`, OpenAI * `insufficient_quota`, …) → rotate. - * 2. Status is not 429 → backoff (caller's domain). + * 2. Status is not a usage-limit status (429/402) → backoff (caller's domain). * 3. Body is absent or {@link isOpaqueStatusBody opaque} (just the status, * empty JSON, HTTP framing only) → rotate conservatively: the server * gave us nothing else to go on. @@ -144,14 +157,15 @@ export function isUsageLimitOutcome(status: number | undefined, message: string } /** - * A 429 body is opaque when it carries no signal beyond the status itself — - * empty, whitespace-only, the status digits with HTTP/JSON framing, or - * generic punctuation. Anything else (retry hints, capacity wording, error - * descriptions) is informative enough to defer to the classifier. + * A usage-limit status body is opaque when it carries no signal beyond the + * status itself — empty, whitespace-only, the status digits with HTTP/JSON + * framing, or generic punctuation. Anything else (retry hints, capacity + * wording, error descriptions) is informative enough to defer to the + * classifier. */ export function isOpaqueStatusBody(message: string): boolean { const cleaned = message - .replace(/\b429\b/g, "") + .replace(/\b(?:429|402)\b/g, "") .replace(/\b(?:http|https|status|error|code|response|message)\b/gi, ""); return !/[a-z\d]{3,}/i.test(cleaned); } @@ -163,5 +177,10 @@ export function isOpaqueStatusBody(message: string): boolean { * {@link isUsageLimitOutcome} uses it for the account-rotation decision. */ export function matchesUsageLimitText(errorMessage: string): boolean { - return USAGE_LIMIT_PATTERN.test(errorMessage) || ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage); + return ( + USAGE_LIMIT_PATTERN.test(errorMessage) || + SPEND_LIMIT_PATTERN.test(errorMessage) || + ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage) || + OPENROUTER_DAILY_FREE_LIMIT_PATTERN.test(errorMessage) + ); } diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index e5599d876..3d2d637b8 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -39,6 +39,8 @@ export * from "./usage/ollama"; export * from "./usage/openai-codex"; export * from "./usage/openai-codex-reset"; export * from "./usage/opencode-go"; +export * from "./usage/synthetic"; +export * from "./usage/xai-oauth"; export * from "./usage/zai"; export * from "./utils/anthropic-auth"; export * from "./utils/event-stream"; diff --git a/packages/ai/src/providers/__tests__/keyless-auth-header.test.ts b/packages/ai/src/providers/__tests__/keyless-auth-header.test.ts new file mode 100644 index 000000000..862803630 --- /dev/null +++ b/packages/ai/src/providers/__tests__/keyless-auth-header.test.ts @@ -0,0 +1,39 @@ +import { describe, expect, test } from "bun:test"; +import { NO_AUTH_SENTINEL, resolveOpenAIRequestSetup } from "../openai-shared"; + +describe("resolveOpenAIRequestSetup keyless auth", () => { + test("omits Authorization for the keyless (auth: none) sentinel, keeping custom headers", () => { + const setup = resolveOpenAIRequestSetup( + { + provider: "qwen", + id: "Qwen3.6-35B-A3B", + baseUrl: "http://localhost:8788", + headers: { "x-api-key": "real-key" }, + }, + { apiKey: NO_AUTH_SENTINEL, messages: [] }, + ); + expect(setup.headers.Authorization).toBeUndefined(); + expect(setup.headers["x-api-key"]).toBe("real-key"); + }); + + test("still injects Bearer for a real key", () => { + const setup = resolveOpenAIRequestSetup( + { provider: "custom", id: "m", baseUrl: "http://localhost:8788" }, + { apiKey: "sk-real", messages: [] }, + ); + expect(setup.headers.Authorization).toBe("Bearer sk-real"); + }); + + test("caller-supplied Authorization in model.headers is preserved even when keyless", () => { + const setup = resolveOpenAIRequestSetup( + { + provider: "qwen", + id: "m", + baseUrl: "http://localhost:8788", + headers: { Authorization: "Bearer custom-token" }, + }, + { apiKey: NO_AUTH_SENTINEL, messages: [] }, + ); + expect(setup.headers.Authorization).toBe("Bearer custom-token"); + }); +}); diff --git a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts index b76cbd6cb..c83d92d82 100644 --- a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts +++ b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts @@ -1,7 +1,13 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, describe, expect, it, vi } from "bun:test"; import { getBundledModel } from "@oh-my-pi/pi-catalog"; -import type { Context } from "../../types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; +import * as kimiOauth from "../../registry/oauth/kimi"; +import { streamSimple } from "../../stream"; +import type { Context, Model } from "../../types"; import type { MessageCreateParamsStreaming } from "../anthropic-wire"; +import { type KimiApiFormat, streamKimi } from "../kimi"; import { streamOpenAIAnthropicShim } from "../openai-anthropic-shim"; import { applyChatCompletionsCompatPolicy, @@ -10,6 +16,15 @@ import { } from "../openai-shared"; const BASE_CHAT_COMPLETIONS_PARAMS: OpenAICompletionsParams = { messages: [], model: "unused", stream: true }; +const KIMI_HEADERS = Object.freeze({ + "User-Agent": "KimiCLI/test", + "X-Msh-Platform": "kimi_cli", + "X-Msh-Version": "test", + "X-Msh-Device-Name": "test", + "X-Msh-Device-Model": "test", + "X-Msh-Os-Version": "test", + "X-Msh-Device-Id": "test", +}); const TITLE_CONTEXT: Context = { systemPrompt: ["Generate a title."], messages: [{ role: "user", content: "Explain the login failure", timestamp: 0 }], @@ -27,8 +42,172 @@ const TITLE_CONTEXT: Context = { ], }; +const K3_MODEL = buildModel({ + id: "k3", + name: "K3", + api: "openai-completions", + provider: "kimi-code", + baseUrl: "https://api.kimi.com/coding/v1", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_048_576, + maxTokens: 32_000, + thinking: { + mode: "effort", + efforts: [Effort.Low, Effort.High, Effort.Max], + defaultLevel: Effort.Max, + requiresEffort: true, + }, + compat: { + thinkingFormat: "kimi", + kimiApiFormat: "openai", + reasoningContentField: "reasoning_content", + supportsDeveloperRole: false, + }, +} satisfies ModelSpec<"openai-completions">); + +async function captureKimiPayload( + model: Model<"openai-completions">, + reasoning: Effort, + format?: KimiApiFormat, +): Promise { + let payload: unknown; + const stream = streamKimi( + model, + { + systemPrompt: [], + messages: [{ role: "user", content: "Reply OK", timestamp: 0 }], + tools: [], + }, + { + ...(format ? { format } : {}), + apiKey: "test-key", + reasoning, + onPayload: body => { + payload = body; + throw new Error("stop after payload capture"); + }, + }, + ); + await stream.result(); + if (payload === undefined) throw new Error("Kimi request payload was not captured"); + return payload; +} + +afterEach(() => { + vi.restoreAllMocks(); +}); + +describe("Kimi K3 thinking transport", () => { + it("sends every live named effort through Kimi's native thinking object by default", async () => { + vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS); + + for (const effort of [Effort.Low, Effort.High, Effort.Max]) { + const payload = await captureKimiPayload(K3_MODEL, effort); + expect(payload).toMatchObject({ thinking: { type: "enabled", effort } }); + expect(payload).not.toHaveProperty("reasoning_effort"); + } + }); + + it("uses adaptive named effort rather than a token budget for an explicit Anthropic override", async () => { + vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS); + + const payload = await captureKimiPayload(K3_MODEL, Effort.Max, "anthropic"); + + expect(payload).toMatchObject({ + thinking: { type: "adaptive" }, + output_config: { effort: Effort.Max }, + }); + expect(payload).not.toHaveProperty("thinking.budget_tokens"); + }); + + it("keeps the legacy K2 default on the Anthropic transport", async () => { + vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS); + const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); + + const payload = await captureKimiPayload(model, Effort.High); + + expect(payload).toMatchObject({ thinking: { type: "enabled" } }); + expect(payload).toHaveProperty("thinking.budget_tokens"); + }); + + it("clamps disabled thinking to the lowest effort for a mandatory-thinking K3", async () => { + vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS); + + let payload: unknown; + const stream = streamSimple( + K3_MODEL, + { + systemPrompt: [], + messages: [{ role: "user", content: "Reply OK", timestamp: 0 }], + tools: [], + }, + { + apiKey: "test-key", + disableReasoning: true, + onPayload: body => { + payload = body; + throw new Error("stop after payload capture"); + }, + }, + ); + await stream.result(); + + expect(payload).toMatchObject({ thinking: { type: "enabled", effort: Effort.Low } }); + expect(payload).not.toMatchObject({ thinking: { type: "disabled" } }); + }); + + it("downgrades named tool choice to required for K3 thinking", async () => { + vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS); + const bundledModel = getBundledModel<"openai-completions">("kimi-code", "k3"); + expect(bundledModel.compat.thinkingFormat).toBe("kimi"); + let payload: unknown; + const capturePayload = async ( + model: Model<"openai-completions">, + toolChoice: "required" | { type: "tool"; name: string }, + tools = TITLE_CONTEXT.tools, + ) => { + const stream = streamKimi( + model, + { ...TITLE_CONTEXT, tools }, + { + apiKey: "test-key", + format: "openai", + reasoning: Effort.Max, + toolChoice, + onPayload: body => { + payload = body; + throw new Error("stop after payload capture"); + }, + }, + ); + await stream.result(); + }; + + for (const model of [K3_MODEL, bundledModel]) { + await capturePayload(model, { type: "tool", name: "set_title" }); + expect(payload).toMatchObject({ + thinking: { type: "enabled" }, + tool_choice: "required", + tools: [{ type: "function", function: { name: "set_title" } }], + }); + + await capturePayload(model, "required"); + expect(payload).toMatchObject({ + thinking: { type: "enabled" }, + tool_choice: "required", + tools: [{ type: "function", function: { name: "set_title" } }], + }); + } + + await capturePayload(K3_MODEL, { type: "tool", name: "missing_tool" }, []); + expect((payload as { tool_choice?: unknown }).tool_choice).toBeUndefined(); + }); +}); + describe("Kimi K2.7 Code thinking policy", () => { - it("omits disabled thinking for title-generator-style Kimi Code requests", () => { + it("expresses disabled thinking explicitly for title-generator-style Kimi Code requests", () => { const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); const policy = resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", @@ -39,11 +218,16 @@ describe("Kimi K2.7 Code thinking policy", () => { applyChatCompletionsCompatPolicy(params, policy); - expect("thinking" in params).toBe(false); - expect(model.compat.supportsForcedToolChoice).toBe(false); + // Kimi's native hosts speak the z.ai binary thinking field: a disabled + // request carries `{ type: "disabled" }` rather than omitting the block. + expect((params as Record).thinking).toEqual({ type: "disabled" }); + // Thinking yields to a forced tool choice (#5758 review): the choice is + // honored and reasoning is turned off, instead of downgrading the choice. + expect(model.compat.supportsForcedToolChoice).toBe(true); + expect(model.compat.disableReasoningOnForcedToolChoice).toBe(true); }); - it("enables thinking and downgrades forced tool choice on Kimi Code's Anthropic endpoint", async () => { + it("keeps the forced tool choice and omits thinking on Kimi Code's Anthropic endpoint", async () => { const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); let payload: MessageCreateParamsStreaming | undefined; const stream = streamOpenAIAnthropicShim( @@ -67,8 +251,43 @@ describe("Kimi K2.7 Code thinking policy", () => { await stream.result(); - expect(payload?.thinking?.type).toBe("enabled"); - expect(payload?.tool_choice).toEqual({ type: "auto" }); + // With reasoning disabled the Anthropic wire carries no thinking block, + // and the forced tool choice survives (thinking yields to the choice). + expect(payload?.thinking).toBeUndefined(); + expect(payload?.tool_choice).toEqual({ type: "tool", name: "set_title" }); + }); + + it("uses the configured Kimi base URL for Anthropic requests", async () => { + vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS); + const bundledModel = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); + const model = { ...bundledModel, baseUrl: "https://gateway.example.com/v1" }; + let requestedUrl: string | undefined; + const stream = streamKimi( + model, + { + systemPrompt: [], + messages: [{ role: "user", content: "Reply OK", timestamp: 0 }], + tools: [], + }, + { + format: "anthropic", + apiKey: "gateway-key", + fetch: async input => { + requestedUrl = String(input); + return new Response( + JSON.stringify({ + type: "error", + error: { type: "authentication_error", message: "stop after URL capture" }, + }), + { status: 401, headers: { "content-type": "application/json" } }, + ); + }, + }, + ); + + await stream.result(); + + expect(requestedUrl).toBe("https://gateway.example.com/v1/messages"); }); it("omits disabled thinking for native Moonshot Kimi K2.7 Code variants", () => { diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 6a3cf1f02..95bf05ce6 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -334,7 +334,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = ( toolConfig, additionalModelRequestFields, }; - options?.onPayload?.(commandInput); + options?.onPayload?.(commandInput, model); const host = `bedrock-runtime.${region}.amazonaws.com`; const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`; diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index a22302fd2..69eef548e 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -100,6 +100,8 @@ export type AnthropicHeaderOptions = { isCloudflareAiGateway?: boolean; claudeCodeSessionId?: string; claudeCodeBetas?: readonly string[]; + /** Allow explicit fingerprint headers to replace OAuth defaults on non-official endpoints. */ + allowAnthropicHeaderOverrides?: boolean; }; export function normalizeAnthropicBaseUrl(baseUrl?: string): string | undefined { @@ -144,7 +146,8 @@ const claudeCodeAgentBetaDefaults = [ midConversationSystemBeta, "advanced-tool-use-2025-11-20", ] as const; -const claudeCodeAgentPostEffortBetas = ["extended-cache-ttl-2025-04-11"] as const; +const extendedCacheTtlBeta = "extended-cache-ttl-2025-04-11"; +const claudeCodeAgentPostEffortBetas = [extendedCacheTtlBeta] as const; const fineGrainedToolStreamingBeta = "fine-grained-tool-streaming-2025-05-14"; const interleavedThinkingBeta = "interleaved-thinking-2025-05-14"; // Asks the API to redact thinking blocks from responses. Only sent when the @@ -225,22 +228,36 @@ export function buildAnthropicHeaders(options: AnthropicHeaderOptions): Record = {}; + const anthropicHeaderOverrides: Record = {}; const filteredEnforcedKeys: string[] = []; - for (const [key, value] of Object.entries(options.modelHeaders ?? {})) { - const lowerKey = key.toLowerCase(); - if (enforcedHeaderKeys.has(lowerKey)) { - // user-agent is always re-applied explicitly. authorization / x-api-key - // are silently re-applied in honoring branches and dropped + logged - // where the branch enforces its own credential. - if (lowerKey === "user-agent") continue; - if (lowerKey === "authorization" && honorAuthorization) continue; - if (lowerKey === "x-api-key" && honorApiKey) continue; - filteredEnforcedKeys.push(key); - continue; + const headerSource = options.modelHeaders; + if (headerSource) { + for (const key in headerSource) { + const value = headerSource[key]; + const lowerKey = key.toLowerCase(); + if (enforcedHeaderKeys.has(lowerKey)) { + if (allowAnthropicHeaderOverrides && overridableAnthropicHeaderKeys.has(lowerKey)) { + anthropicHeaderOverrides[key] = value; + continue; + } + // user-agent is always re-applied explicitly. authorization / x-api-key + // are silently re-applied in honoring branches and dropped + logged + // where the branch enforces its own credential. + if (lowerKey === "user-agent") continue; + if (lowerKey === "authorization" && honorAuthorization) continue; + if (lowerKey === "x-api-key" && honorApiKey) continue; + filteredEnforcedKeys.push(key); + continue; + } + modelHeaders[key] = value; } - modelHeaders[key] = value; } if (filteredEnforcedKeys.length > 0) { // Caller/env-supplied values (options.headers, ANTHROPIC_CUSTOM_HEADERS) @@ -266,7 +283,7 @@ export function buildAnthropicHeaders(options: AnthropicHeaderOptions): Record key.toLowerCase()), ); +const overridableAnthropicHeaderKeys = new Set( + [...Object.keys(claudeCodeHeaders), "anthropic-beta", "User-Agent", "x-app"].map(key => key.toLowerCase()), +); + const CLAUDE_BILLING_HEADER_PREFIX = "x-anthropic-billing-header:"; function createClaudeBillingHeader(firstUserMessageText: string): string { @@ -1167,6 +1197,25 @@ function resolveAnthropicBaseUrl(model: Model<"anthropic-messages">, apiKey?: st return normalizeAnthropicBaseUrl(model.baseUrl); } +function resolveEagerToolInputStreamingSupport( + model: Model<"anthropic-messages">, + effectiveBaseUrl: string | undefined, +): boolean { + if (!model.compat.supportsEagerToolInputStreaming) return false; + // First-party Anthropic endpoints accept the per-tool flag. + if (isOfficialAnthropicApiUrl(effectiveBaseUrl)) return true; + // Non-official effective endpoint. `supportsEagerToolInputStreaming` may be + // stale-true here because compat is materialized once at build time and is + // never rebuilt for a baseUrl-only reroute — either a runtime provider + // override (`pi.registerProvider("anthropic", { baseUrl })`) or Foundry + // (`CLAUDE_CODE_USE_FOUNDRY`). Both leave the canonical model's resolved + // compat in place. `officialEndpoint` records whether compat was built for + // the canonical Anthropic URL, so only endpoints whose compat was authored + // for a non-official host (an explicit `compat.supportsEagerToolInputStreaming` + // opt-in on a custom `baseUrl`) still send the field. + return !model.compat.officialEndpoint; +} + function parseAnthropicCustomHeaders(rawHeaders: string | undefined): Record | undefined { const source = rawHeaders?.trim(); if (!source) return undefined; @@ -1741,6 +1790,7 @@ const streamAnthropicOnce = ( } const apiKey = options?.apiKey ?? getEnvApiKey(model.provider) ?? ""; const baseUrl = resolveAnthropicBaseUrl(model, apiKey) ?? "https://api.anthropic.com"; + const supportsEagerToolInputStreaming = resolveEagerToolInputStreamingSupport(model, baseUrl); const providerSessionState = getAnthropicProviderSessionState( options?.providerSessionState, baseUrl, @@ -1752,6 +1802,23 @@ const streamAnthropicOnce = ( let forceDemoteUnsignedThinking = providerSessionState?.replayUnsignedThinkingDisabled ?? false; const mergedCallerHeaders = mergeHeaders(model.headers, options?.headers); const umansGatewayWebSearchHeader = getUmansWebSearchHeader(model, mergedCallerHeaders); + // Keep fallback payloads aligned with the top-level Vertex effort gate: + // no nested effort field means the fallback scan cannot re-add its beta. + let fallbacks = options?.fallbacks; + if ( + model.provider === "google-vertex" && + fallbacks?.some(entry => entry.output_config?.effort !== undefined) + ) { + fallbacks = fallbacks.map(entry => { + const outputConfig = entry.output_config; + if (outputConfig?.effort === undefined) return entry; + return { + ...entry, + output_config: + outputConfig.task_budget === undefined ? undefined : { task_budget: outputConfig.task_budget }, + }; + }); + } let client: AnthropicMessagesClientLike; let isOAuthToken: boolean; @@ -1776,6 +1843,10 @@ const streamAnthropicOnce = ( // the toggle cannot 400); the beta must accompany the field in both. // MiniMax uses `thinking.type:"adaptive"` itself as the control surface, // so the sentinel "adaptive" value intentionally sends no output_config. + // Skip Vertex rawPredict: that adapter needs betas in the body + // (`anthropic_beta`), not as an `anthropic-beta` HTTP header, so the + // effort field is dropped from the body there too (see buildParams) and + // advertising the beta would only earn a 400 (#5614). const sendsAdaptiveEffortPin = options?.thinkingEnabled === false && model.thinking?.mode === "anthropic-adaptive" && @@ -1783,6 +1854,7 @@ const streamAnthropicOnce = ( !usesAdaptiveThinkingTagOnly(model); if ( model.reasoning && + model.provider !== "google-vertex" && ((options?.thinkingEnabled && options.effort !== "adaptive") || sendsAdaptiveEffortPin) && !extraBetas.includes(effortBeta) ) { @@ -1811,16 +1883,27 @@ const streamAnthropicOnce = ( ) { extraBetas.push(contextManagementBeta); } + // `ttl: "1h"` requires the extended-cache-ttl beta on API-key + // requests. OAuth requests never add it here: agent requests + // already carry it in the Claude Code beta list, and utility + // requests must not deviate from CC's header fingerprint. + if ( + !(options?.isOAuth ?? isAnthropicOAuthToken(apiKey)) && + getCacheControl(model, options?.cacheRetention, false).cacheControl?.ttl === "1h" && + !extraBetas.includes(extendedCacheTtlBeta) + ) { + extraBetas.push(extendedCacheTtlBeta); + } // Server-side fallback beta chain: opt-in via `options.fallbacks`. // Nested overrides (`speed`, `output_config.effort`, // `output_config.task_budget`) reuse the same top-level betas // Anthropic requires for the primary request, so scan the chain // and add every companion beta the fallback entries touch. - if (options?.fallbacks?.length) { + if (fallbacks?.length) { if (!extraBetas.includes(serverSideFallbackBeta)) { extraBetas.push(serverSideFallbackBeta); } - for (const entry of options.fallbacks) { + for (const entry of fallbacks) { if (entry.speed === "fast" && !extraBetas.includes(fastModeBeta)) { extraBetas.push(fastModeBeta); } @@ -1854,15 +1937,13 @@ const streamAnthropicOnce = ( } const preparedContext = await prepareAnthropicManyImageContext(context, model.input.includes("image")); const prepareParams = async (): Promise => { - let nextParams = buildParams( - model, - preparedContext, - isOAuthToken, - options, + let nextParams = buildParams(model, preparedContext, isOAuthToken, options, { disableStrictTools, - umansGatewayWebSearchHeader !== undefined, + useUmansGatewayWebSearch: umansGatewayWebSearchHeader !== undefined, forceDemoteUnsignedThinking, - ); + supportsEagerToolInputStreaming, + fallbacks, + }); if (disableStrictTools) { dropAnthropicStrictTools(nextParams); } @@ -1888,9 +1969,9 @@ const streamAnthropicOnce = ( // Opt-in flag: the response parser only honors `fallback` content // blocks and `usage.iterations` when the current request opted into - // the server-side-fallback beta chain. Leaving `options.fallbacks` - // unset preserves the pre-fallback stream shape on every event. - const serverSideFallback = !!options?.fallbacks?.length; + // server-side-fallback beta chain. Leaving `fallbacks` unset preserves + // the pre-fallback stream shape on every event. + const serverSideFallback = !!fallbacks?.length; type Block = ( | ThinkingContent | RedactedThinkingContent @@ -2682,9 +2763,11 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A const compat = model.compat; const disableStrictTools = disableStrictToolsOverride ?? compat.disableStrictTools; const needsInterleavedBeta = interleavedThinking && !model.thinking?.supportsDisplay; - const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming; const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey); const baseUrl = resolveAnthropicBaseUrl(model, apiKey); + const supportsEagerToolInputStreaming = resolveEagerToolInputStreamingSupport(model, baseUrl); + const needsFineGrainedToolStreamingBeta = + hasTools && isOfficialAnthropicApiUrl(baseUrl) && !supportsEagerToolInputStreaming; const foundryCustomHeaders = resolveAnthropicCustomHeaders(model); const tlsFetchOptions = buildClaudeCodeTlsFetchOptions(model, baseUrl); // Disable Bun's native ~300s pre-response fetch timeout (issue #2422). @@ -2699,9 +2782,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A if (model.provider === "github-copilot") { const copilotApiKey = parseGitHubCopilotApiKey(apiKey).accessToken; // The GitHub Copilot Anthropic proxy doesn't accept Anthropic beta - // features (and the catalog already forces `supportsEagerToolInputStreaming - // = false` for this host, so `needsFineGrainedToolStreamingBeta` is true - // whenever tools are present). Forward only caller-supplied betas. + // features. Forward only caller-supplied betas. const betaFeatures = [...extraBetas]; const defaultHeaders = mergeHeaders( { @@ -2751,6 +2832,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A dynamicHeaders, ), isCloudflareAiGateway: model.provider === "cloudflare-ai-gateway", + allowAnthropicHeaderOverrides: model.compat.allowAnthropicHeaderOverrides, claudeCodeSessionId, claudeCodeBetas: oauthToken ? buildClaudeCodeBetas( @@ -3129,15 +3211,29 @@ function extractClaudeCodeFirstUserMessageText(messages: readonly Message[]): st return ""; } +type AnthropicParamBuildOptions = { + disableStrictTools: boolean; + useUmansGatewayWebSearch: boolean; + forceDemoteUnsignedThinking: boolean; + supportsEagerToolInputStreaming: boolean; + /** Sanitized server-side fallback entries; defaults to `options?.fallbacks` when omitted. */ + fallbacks?: AnthropicOptions["fallbacks"]; +}; + function buildParams( model: Model<"anthropic-messages">, context: Context, isOAuthToken: boolean, - options?: AnthropicOptions, - disableStrictTools = false, - useUmansGatewayWebSearch = false, - forceDemoteUnsignedThinking = false, + options: AnthropicOptions | undefined, + buildOptions: AnthropicParamBuildOptions, ): MessageCreateParamsStreaming { + const { + disableStrictTools, + useUmansGatewayWebSearch, + forceDemoteUnsignedThinking, + supportsEagerToolInputStreaming, + fallbacks = options?.fallbacks, + } = buildOptions; // A session-scoped auto-demote (learned from a live signing 400) clones the // resolved compat with `replayUnsignedThinking: false` so every subsequent // downstream read (convertAnthropicMessages, transformMessages) sees the @@ -3165,7 +3261,7 @@ function buildParams( context.tools, isOAuthToken, disableStrictTools || model.provider === "github-copilot", - model.compat.supportsEagerToolInputStreaming, + supportsEagerToolInputStreaming, model.compat.escapeBuiltinToolNames, useUmansGatewayWebSearch, ); @@ -3255,9 +3351,12 @@ function buildParams( ? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] } : undefined; - // Pre-compute output_config. + // Pre-compute output_config. Skip `effort` on Vertex rawPredict: it requires + // the `effort-2025-11-24` beta, which that adapter can only accept in the body + // (`anthropic_beta`), never as the `anthropic-beta` HTTP header this path sets + // — so the field is dropped alongside the beta to avoid a 400 (#5614). const outputConfigEntries: AnthropicOutputConfig = {}; - if (outputConfigEffort) outputConfigEntries.effort = outputConfigEffort; + if (outputConfigEffort && model.provider !== "google-vertex") outputConfigEntries.effort = outputConfigEffort; if (options?.taskBudget) outputConfigEntries.task_budget = options.taskBudget; const outputConfig = Object.keys(outputConfigEntries).length ? outputConfigEntries : undefined; @@ -3272,7 +3371,7 @@ function buildParams( const params: MessageCreateParamsStreaming = { model: options?.requestModelId ?? model.requestModelId ?? model.id, messages: convertAnthropicMessages(context.messages, effectiveModel, isOAuthToken, { - serverSideFallbackEnabled: !!options?.fallbacks?.length, + serverSideFallbackEnabled: !!fallbacks?.length, }), ...(systemBlocks && { system: systemBlocks }), ...(tools !== undefined && { tools }), @@ -3281,7 +3380,7 @@ function buildParams( ...(thinking && { thinking }), ...(contextManagement && { context_management: contextManagement }), ...(outputConfig && { output_config: outputConfig }), - ...(options?.fallbacks?.length ? { fallbacks: options.fallbacks } : {}), + ...(fallbacks?.length ? { fallbacks } : {}), stream: true, }; diff --git a/packages/ai/src/providers/cursor.ts b/packages/ai/src/providers/cursor.ts index e259d62b1..b1070c609 100644 --- a/packages/ai/src/providers/cursor.ts +++ b/packages/ai/src/providers/cursor.ts @@ -213,6 +213,34 @@ function parseConnectEndStream(data: Uint8Array): Error | null { } } +/** + * Maps an opaque HTTP/2 negotiation failure into an actionable error. + * + * bun only opens an HTTP/2 session when TLS-ALPN negotiates `h2`. Behind a + * TLS-intercepting proxy that strips ALPN (e.g. Zscaler), the handshake yields + * no `h2` protocol and bun throws `ERR_HTTP2_ERROR: h2 is not supported`. The + * Cursor run RPC is HTTP/2-only (the ALB rejects HTTP/1.1 with 464), so there + * is no h1 fallback the way model discovery has one — the run simply cannot + * proceed. Replace the opaque message with one that names the cause and points + * at the `providers.cursor.baseUrl` workaround. + * + * Non-ALPN errors pass through untouched. + */ +export function mapH2TransportError(error: unknown, baseUrl: string): unknown { + const code = (error as { code?: unknown } | null)?.code; + const message = error instanceof Error ? error.message : String(error); + if (code === "ERR_HTTP2_ERROR" && /h2 is not supported/i.test(message)) { + return new AIError.ProviderResponseError( + `Cursor run transport could not negotiate HTTP/2 with ${baseUrl}: "h2 is not supported". ` + + "This host serves the run RPC over HTTP/2 only, and the TLS handshake did not negotiate " + + "h2 via ALPN — typically an ALPN-stripping TLS-intercepting proxy (e.g. Zscaler). " + + "Front the provider with a local HTTP/2 bridge and set providers.cursor.baseUrl to it.", + { provider: "cursor", kind: "runtime", cause: error }, + ); + } + return error; +} + function debugBytes(bytes: Uint8Array, asHex: boolean): string { if (asHex) { return Buffer.from(bytes).toString("hex"); @@ -352,7 +380,30 @@ export const streamCursor: StreamFunction<"cursor-agent"> = ( let heartbeatTimer: NodeJS.Timeout | null = null; let debugResponseLogPromise: Promise | undefined; const h2Completion = Promise.withResolvers(); - let resolveH2: (() => void) | undefined = h2Completion.resolve; + let h2Settled = false; + let sawTurnEnded = false; + let endStreamError: Error | null = null; + const settleH2 = (error?: unknown): void => { + if (h2Settled) return; + h2Settled = true; + if (error !== undefined) { + h2Completion.reject(error); + return; + } + if (endStreamError) { + h2Completion.reject(endStreamError); + return; + } + if (!sawTurnEnded) { + h2Completion.reject( + new AIError.ProviderResponseError("Cursor stream ended before turnEnded", { + kind: "incomplete-stream", + }), + ); + return; + } + h2Completion.resolve(); + }; try { const apiKey = options?.apiKey; @@ -408,17 +459,17 @@ export const streamCursor: StreamFunction<"cursor-agent"> = ( } else { h2Client = http2.connect(baseUrl); } - h2Client.on("error", h2Completion.reject); + h2Client.on("error", error => settleH2(mapH2TransportError(error, baseUrl))); h2Request = h2Client.request(requestHeaders); stream.push({ type: "start", partial: output }); let pendingBuffer = Buffer.alloc(0); - let endStreamError: Error | null = null; let currentTextBlock: (TextContent & { [kStreamingBlockIndex]: number }) | null = null; let currentThinkingBlock: (ThinkingContent & { [kStreamingBlockIndex]: number }) | null = null; let currentToolCall: ToolCallState | null = null; + const resolvedMcpToolCallIds = new Set(); const usageState: UsageState = { sawTokenDelta: false }; const state: BlockState = { @@ -431,6 +482,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = ( get currentToolCall() { return currentToolCall; }, + resolvedMcpToolCallIds, get firstTokenTime() { return firstTokenTime; }, @@ -505,12 +557,9 @@ export const streamCursor: StreamFunction<"cursor-agent"> = ( log("error", "handleServerMessage", { error: String(error) }); }); - // Resolve only on explicit turnEnded. stopReason defaults to "stop" - // and is not a reliable signal for stream completion. - if (isTurnEnded && resolveH2) { - const r = resolveH2; - resolveH2 = undefined; - r(); + // Application completion is not protocol success; wait for a clean HTTP/2 end. + if (isTurnEnded) { + sawTurnEnded = true; } } catch (e) { log("error", "parseServerMessage", { error: String(e) }); @@ -537,40 +586,30 @@ export const streamCursor: StreamFunction<"cursor-agent"> = ( h2Request.on("trailers", trailers => { const status = trailers["grpc-status"]; const msg = trailers["grpc-message"]; - if (status && status !== "0") { - void closeDebugLog().finally(() => { - h2Completion.reject( - new AIError.ProviderResponseError( - `gRPC error ${status}: ${decodeURIComponent(String(msg || ""))}`, - { kind: "envelope" }, - ), - ); - }); + if (status && status !== "0" && !endStreamError) { + endStreamError = new AIError.ProviderResponseError( + `gRPC error ${status}: ${decodeURIComponent(String(msg || ""))}`, + { kind: "envelope" }, + ); } }); h2Request.on("end", () => { - resolveH2 = undefined; void closeDebugLog() - .then(() => { - if (endStreamError) { - h2Completion.reject(endStreamError); - return; - } - h2Completion.resolve(); - }) - .catch(h2Completion.reject); + .then(() => settleH2()) + .catch(error => settleH2(error)); }); h2Request.on("error", error => { - void closeDebugLog().finally(() => h2Completion.reject(error)); + const mapped = mapH2TransportError(error, baseUrl); + void closeDebugLog().finally(() => settleH2(mapped)); }); if (options?.signal) { options.signal.addEventListener("abort", () => { h2Request?.close(); void closeDebugLog().finally(() => { - h2Completion.reject(new AIError.AbortError()); + settleH2(new AIError.AbortError()); }); }); } @@ -640,6 +679,8 @@ export interface BlockState { currentTextBlock: (TextContent & { [kStreamingBlockIndex]: number }) | null; currentThinkingBlock: (ThinkingContent & { [kStreamingBlockIndex]: number }) | null; currentToolCall: ToolCallState | null; + /** MCP call IDs executed through Cursor's exec channel before their stream block arrives. */ + resolvedMcpToolCallIds: Set; firstTokenTime: number | undefined; setTextBlock: (b: (TextContent & { [kStreamingBlockIndex]: number }) | null) => void; setThinkingBlock: (b: (ThinkingContent & { [kStreamingBlockIndex]: number }) | null) => void; @@ -1292,6 +1333,13 @@ async function handleExecServerMessage( case "mcpArgs": { const args = execMsg.message.value; const mcpCall = decodeMcpCall(args); + if (execHandlers?.mcp) { + if (state.currentToolCall?.id === mcpCall.toolCallId) { + state.currentToolCall[kCursorExecResolved] = true; + } else { + state.resolvedMcpToolCallIds.add(mcpCall.toolCallId); + } + } const { execResult } = await resolveExecHandler( mcpCall, execHandlers?.mcp?.bind(execHandlers), @@ -2217,15 +2265,19 @@ export function processInteractionUpdate( const mcpCall = toolCall.mcpToolCall; if (mcpCall) { const args = mcpCall.args || {}; + const id = args.toolCallId || crypto.randomUUID(); const block: ToolCallState = { type: "toolCall", - id: args.toolCallId || crypto.randomUUID(), + id, name: args.name || args.toolName || "", arguments: {}, [kStreamingBlockIndex]: output.content.length, [kStreamingPartialJson]: "", [kStreamingBlockKind]: "mcp", }; + if (state.resolvedMcpToolCallIds.delete(id)) { + block[kCursorExecResolved] = true; + } output.content.push(block); state.setToolCall(block); stream.push({ type: "toolcall_start", contentIndex: output.content.length - 1, partial: output }); @@ -2816,7 +2868,7 @@ function buildGrpcRequest( conversationId: state.conversationId, }); - options?.onPayload?.(runRequest); + options?.onPayload?.(runRequest, model); // Tools are sent later via requestContext (exec handshake) diff --git a/packages/ai/src/providers/devin.ts b/packages/ai/src/providers/devin.ts index c3f546e0b..7b5113b81 100644 --- a/packages/ai/src/providers/devin.ts +++ b/packages/ai/src/providers/devin.ts @@ -43,9 +43,11 @@ import type { Tool, ToolCall, } from "../types"; +import { isDemotedThinking } from "../utils/block-symbols"; import { deterministicUuid } from "../utils/deterministic-id"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { toolWireSchema } from "../utils/schema/wire"; +import { transformMessages } from "./transform-messages"; /** Base host for Codeium/Windsurf's Cascade chat API (Connect protocol over HTTP/1.1). */ export const DEVIN_API_URL = "https://server.codeium.com"; @@ -78,6 +80,13 @@ const CONNECT_END_STREAM_FLAG = 0x02; * fails fast instead of consuming memory. */ const MAX_CONNECT_FRAME_PAYLOAD = 16 * 1024 * 1024; +/** + * Recovery heuristic for opaque Devin `invalid_argument` trailers. This is not + * asserted to be the backend's hard limit: small requests can hit the same + * intermittent error, while compactable message history this large is likely + * to benefit from the existing context-overflow maintenance path. + */ +const LARGE_HISTORY_RECOVERY_BYTES = 512 * 1024; export const streamDevin: StreamFunction<"devin-agent"> = ( model: Model<"devin-agent">, @@ -157,10 +166,14 @@ export const streamDevin: StreamFunction<"devin-agent"> = ( const auth = await fetchDevinAuthMetadata(apiKey, baseUrl, fetchImpl, options?.signal); const chatBaseUrl = auth.baseUrl ?? baseUrl; const request = buildDevinChatRequest(model, context, options, apiKey, auth.userJwt); - logger.debug("devin: sending chat request", { model: model.id, tools: context.tools?.length ?? 0 }); - const reqBytes = toBinary(GetChatMessageRequestSchema, request); const gz = gzipSync(reqBytes); + logger.debug("devin: sending chat request", { + model: model.id, + tools: context.tools?.length ?? 0, + requestBytes: reqBytes.byteLength, + compressedBytes: gz.byteLength, + }); const frame = Buffer.alloc(5 + gz.length); frame[0] = CONNECT_COMPRESSED_FLAG; frame.writeUInt32BE(gz.length, 1); @@ -223,7 +236,53 @@ export const streamDevin: StreamFunction<"devin-agent"> = ( if (flag & CONNECT_END_STREAM_FLAG) { const trailerBytes = flag & CONNECT_COMPRESSED_FLAG ? gunzipSync(payload) : payload; const trailerError = readConnectTrailerError(trailerBytes.toString("utf8").trim()); - if (trailerError) throw new AIError.ValidationError(trailerError); + if (trailerError) { + const error = new AIError.ValidationError(trailerError.formatted); + if ( + firstTokenTime === undefined && + trailerError.code.toLowerCase() === "invalid_argument" && + /\binternal error\b/i.test(trailerError.message) + ) { + // The full protobuf also contains the system prompt and tool + // schemas, which history maintenance cannot shrink. Re-encode + // only the repeated history field before choosing recovery. + let activeTailCount = 0; + const lastRole = context.messages.at(-1)?.role; + if (lastRole === "user" || lastRole === "developer") { + activeTailCount = 1; + // A trailing developer message can accompany the current user + // prompt. Earlier user-role records may instead be flushed + // execution history and must remain eligible for compaction. + if (lastRole === "developer") { + for (let i = context.messages.length - 2; i >= 0; i--) { + const role = context.messages[i].role; + if (role !== "user" && role !== "developer") break; + activeTailCount++; + } + } + } + const shrinkablePrompts = + activeTailCount > 0 + ? request.chatMessagePrompts.slice(0, -activeTailCount) + : request.chatMessagePrompts; + const historyBytes = toBinary( + GetChatMessageRequestSchema, + create(GetChatMessageRequestSchema, { + chatMessagePrompts: shrinkablePrompts, + }), + ).byteLength; + if (historyBytes >= LARGE_HISTORY_RECOVERY_BYTES) { + AIError.attach(error, AIError.create(AIError.Flag.ContextOverflow)); + logger.warn("devin: treating large-history invalid_argument as context overflow", { + model: model.id, + historyBytes, + requestBytes: reqBytes.byteLength, + compressedBytes: gz.byteLength, + }); + } + } + throw error; + } continue; } @@ -322,7 +381,8 @@ export const streamDevin: StreamFunction<"devin-agent"> = ( output.usage.output = Number(msg.usage.outputTokens); output.usage.cacheRead = Number(msg.usage.cacheReadTokens); output.usage.cacheWrite = Number(msg.usage.cacheWriteTokens); - output.usage.totalTokens = output.usage.input + output.usage.output; + output.usage.totalTokens = + output.usage.input + output.usage.output + output.usage.cacheRead + output.usage.cacheWrite; } } @@ -442,6 +502,7 @@ function buildDevinChatRequest( options?.stopSequences && options.stopSequences.length > 0 ? [...DEVIN_DEFAULT_STOP_PATTERNS, ...options.stopSequences] : DEVIN_DEFAULT_STOP_PATTERNS; + const messages = transformMessages(context.messages, model); return create(GetChatMessageRequestSchema, { metadata: create(MetadataSchema, { apiKey, @@ -453,7 +514,7 @@ function buildDevinChatRequest( locale: "en", }), prompt: (context.systemPrompt ?? []).join("\n\n"), - chatMessagePrompts: buildChatMessagePrompts(context.messages, cascadeId), + chatMessagePrompts: buildChatMessagePrompts(messages, cascadeId, model), chatModelUid: options?.chatModelUid ?? model.requestModelId ?? model.id, requestType: ChatMessageRequestType.CASCADE, plannerMode: ConversationalPlannerMode.DEFAULT, @@ -485,7 +546,11 @@ function buildDevinChatRequest( } /** Map omp `Message` history onto Cascade `ChatMessagePrompt`s (USER / SYSTEM / TOOL channels). */ -function buildChatMessagePrompts(messages: Message[], cascadeId: string): ChatMessagePrompt[] { +function buildChatMessagePrompts( + messages: Message[], + cascadeId: string, + model: Model<"devin-agent">, +): ChatMessagePrompt[] { const prompts: ChatMessagePrompt[] = []; // messageId seeds are `cascadeId\0index\0role[...]` — prompt text is excluded // so ids stay stable across content edits / history rebuilds. @@ -513,16 +578,18 @@ function buildChatMessagePrompts(messages: Message[], cascadeId: string): ChatMe }), ); } else if (msg.role === "assistant") { + const isNativeDevinMessage = + msg.api === model.api && msg.provider === model.provider && msg.model === model.id; let promptText = ""; let thinkingText = ""; let signature = ""; const toolCalls: ChatToolCall[] = []; for (const part of msg.content) { if (part.type === "text") { - promptText += part.text; + promptText += `${part.text}${isDemotedThinking(part) ? "\n" : ""}`; } else if (part.type === "thinking") { thinkingText += part.thinking; - if (!signature && part.thinkingSignature) signature = part.thinkingSignature; + if (isNativeDevinMessage && !signature && part.thinkingSignature) signature = part.thinkingSignature; } else if (part.type === "toolCall") { toolCalls.push( create(ChatToolCallSchema, { @@ -533,9 +600,13 @@ function buildChatMessagePrompts(messages: Message[], cascadeId: string): ChatMe ); } } + if (!promptText && !thinkingText && !signature && toolCalls.length === 0) continue; prompts.push( create(ChatMessagePromptSchema, { - messageId: msg.responseId ?? `bot-${deterministicUuid(`${cascadeId}\0${index}\0assistant`)}`, + messageId: + isNativeDevinMessage && msg.responseId + ? msg.responseId + : `bot-${deterministicUuid(`${cascadeId}\0${index}\0assistant`)}`, source: ChatMessageSource.SYSTEM, prompt: promptText, thinking: thinkingText, @@ -569,12 +640,18 @@ function buildChatMessagePrompts(messages: Message[], cascadeId: string): ChatMe return prompts; } +interface ConnectTrailerError { + code: string; + message: string; + formatted: string; +} + /** - * Parse a Connect end-of-stream JSON trailer and return a human-readable error - * string when it carries `{ error: { code, message } }`, else `null`. The trailer - * is untrusted server output, so the shape is checked with guards rather than asserted. + * Parse a Connect end-of-stream JSON trailer and return its structured error + * when it carries `{ error: { code, message } }`, else `null`. The trailer is + * untrusted server output, so the shape is checked with guards rather than asserted. */ -function readConnectTrailerError(text: string): string | null { +function readConnectTrailerError(text: string): ConnectTrailerError | null { if (text.length === 0) return null; let parsed: unknown; try { @@ -588,5 +665,9 @@ function readConnectTrailerError(text: string): string | null { const code = "code" in err && typeof err.code === "string" ? err.code : ""; const message = "message" in err && typeof err.message === "string" ? err.message : ""; if (!code && !message) return null; - return `Devin stream error${code ? ` ${code}` : ""}: ${message}`; + return { + code, + message, + formatted: `Devin stream error${code ? ` ${code}` : ""}: ${message}`, + }; } diff --git a/packages/ai/src/providers/gitlab-duo-workflow.ts b/packages/ai/src/providers/gitlab-duo-workflow.ts index 6f1a66928..e9f197e13 100644 --- a/packages/ai/src/providers/gitlab-duo-workflow.ts +++ b/packages/ai/src/providers/gitlab-duo-workflow.ts @@ -24,6 +24,7 @@ import { normalizeSystemPrompts } from "../utils"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { toolWireSchema } from "../utils/schema/wire"; import chatmlHistoryNote from "./gitlab-duo-workflow-chatml-note.md" with { type: "text" }; +import { redactSensitiveCredentials } from "./transform-messages"; export const GITLAB_DUO_WORKFLOW_PROVIDER_ID = "gitlab-duo-agent"; export const GITLAB_DUO_WORKFLOW_API = "gitlab-duo-agent"; @@ -2581,10 +2582,13 @@ function isGitLabDuoWorkflowChatMlGoal(context: Context): boolean { // conversation sequences the way `Human:`/`Assistant:` are. function buildGitLabDuoWorkflowGoal(context: Context): string { const conversation = buildGitLabDuoWorkflowConversationHistory(context.messages); + // The goal transcript bypasses transformMessages, so apply the outbound + // credential redaction here — the same scrub the flow-config system slot + // already receives — before the payload leaves the process. if (conversation.length <= 1) { - return extractLatestUserPrompt(context.messages); + return redactSensitiveCredentials(extractLatestUserPrompt(context.messages)); } - return renderGitLabDuoWorkflowChatMl(conversation); + return redactSensitiveCredentials(renderGitLabDuoWorkflowChatMl(conversation)); } const GITLAB_DUO_WORKFLOW_CHATML_START = "<|im_start|>"; diff --git a/packages/ai/src/providers/google-gemini-cli.ts b/packages/ai/src/providers/google-gemini-cli.ts index ee112e4bc..b1fc40508 100644 --- a/packages/ai/src/providers/google-gemini-cli.ts +++ b/packages/ai/src/providers/google-gemini-cli.ts @@ -815,9 +815,6 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( if (isBuffering) { const buffered = consumePlanningBuffer(textBuffer, toolNames); if (buffered.kind !== "incomplete") { - if (buffered.kind === "leak") { - sawLeak = true; - } const visibleSignature = bufferedTextSignature; isBuffering = false; textBuffer = ""; @@ -896,9 +893,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( if (isBuffering && textBuffer !== "") { const buffered = consumePlanningBuffer(textBuffer, toolNames, true); - if (buffered.kind === "leak") { - sawLeak = true; - } + if (buffered.kind !== "incomplete") { feedVisibleText(buffered.visibleText, bufferedTextSignature); } @@ -910,11 +905,10 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( flushVisibleText(bufferedTextSignature); endCurrentBlock(); - return hasMeaningfulGoogleContent(output) || sawLeak; + return hasMeaningfulGoogleContent(output); }; let receivedContent = false; - let sawLeak = false; for (let i = 0; i < endpoints.length; i++) { const endpoint = endpoints[i]; @@ -1055,10 +1049,15 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( break; } catch (error) { const status = extractHttpStatusFromError(error); - if (AIError.isTransientStatus(status)) { - if (!isLastEndpoint && !started) { - continue; - } + if ( + !isLastEndpoint && + !started && + (AIError.isTransientStatus(status) || + (status === undefined && + !(error instanceof AIError.ProviderResponseError && error.kind === "output") && + AIError.retriable(AIError.classify(error)))) + ) { + continue; } throw error; } diff --git a/packages/ai/src/providers/kimi.ts b/packages/ai/src/providers/kimi.ts index 276fe3da4..82fadeddd 100644 --- a/packages/ai/src/providers/kimi.ts +++ b/packages/ai/src/providers/kimi.ts @@ -5,8 +5,8 @@ * - OpenAI: https://api.kimi.com/coding/v1/chat/completions * - Anthropic: https://api.kimi.com/coding/v1/messages * - * The Anthropic API is generally more stable and recommended. - * Note: Kimi calculates TPM rate limits based on max_tokens, not actual output. + * Each discovered model selects its server-declared protocol; legacy models + * without protocol metadata retain the Anthropic-compatible default. */ import { getKimiCommonHeaders } from "../registry/oauth/kimi"; @@ -20,11 +20,8 @@ import { export type KimiApiFormat = OpenAIAnthropicApiFormat; -// Note: Anthropic SDK appends /v1/messages, so base URL should not include /v1 -const KIMI_ANTHROPIC_BASE_URL = "https://api.kimi.com/coding"; - export interface KimiOptions extends OpenAIAnthropicShimOptions { - /** API format: "openai" or "anthropic". Default: "anthropic" */ + /** Explicit API format override. Defaults to the model's discovered protocol. */ format?: KimiApiFormat; } @@ -38,8 +35,9 @@ export function streamKimi( options?: KimiOptions, ): AssistantMessageEventStream { return streamOpenAIAnthropicShim(model, context, options, { - anthropicBaseUrl: KIMI_ANTHROPIC_BASE_URL, - defaultFormat: "anthropic", + anthropicBaseUrl: model.baseUrl.replace(/\/v1\/?$/, ""), + defaultFormat: model.compat.kimiApiFormat ?? "anthropic", + anthropicThinkingMode: model.compat.thinkingFormat === "kimi" ? "anthropic-adaptive" : undefined, extraHeaders: getKimiCommonHeaders, }); } diff --git a/packages/ai/src/providers/openai-anthropic-shim.ts b/packages/ai/src/providers/openai-anthropic-shim.ts index 8fdfa374e..4b3372e20 100644 --- a/packages/ai/src/providers/openai-anthropic-shim.ts +++ b/packages/ai/src/providers/openai-anthropic-shim.ts @@ -10,7 +10,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { ANTHROPIC_THINKING, mapAnthropicToolChoice } from "../stream"; -import type { Context, Model, ModelSpec, SimpleStreamOptions } from "../types"; +import type { Context, Model, ModelSpec, SimpleStreamOptions, ThinkingControlMode } from "../types"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { createProviderErrorMessage } from "./error-message"; import { streamAnthropic, streamOpenAICompletions } from "./register-builtins"; @@ -29,6 +29,8 @@ export interface OpenAIAnthropicShimConfig { openaiBaseUrl?: string; /** Default API format when caller does not specify one. */ defaultFormat: OpenAIAnthropicApiFormat; + /** Thinking transport used when this provider's Anthropic endpoint differs from generic budget semantics. */ + anthropicThinkingMode?: ThinkingControlMode; /** Provider-specific headers (e.g. auth/session) merged ahead of user-supplied headers. */ extraHeaders?: () => Record; } @@ -67,6 +69,9 @@ export function streamOpenAIAnthropicShim( contextWindow: model.contextWindow, maxTokens: model.maxTokens, reasoning: model.reasoning, + ...(config.anthropicThinkingMode && model.thinking + ? { thinking: { ...model.thinking, mode: config.anthropicThinkingMode } } + : {}), input: model.input, cost: model.cost, } as ModelSpec<"anthropic-messages">); @@ -95,6 +100,7 @@ export function streamOpenAIAnthropicShim( fetch: options?.fetch, thinkingEnabled, thinkingBudgetTokens: thinkingBudget, + reasoning: config.anthropicThinkingMode ? reasoningEffort : undefined, toolChoice: mapAnthropicToolChoice(options?.toolChoice), serviceTier: options?.serviceTier, }); diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 4a6b14d4a..0a582cf0a 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -60,6 +60,7 @@ import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout, } from "../utils/idle-iterator"; +import { getProxyForProvider, shouldBypassProxy } from "../utils/proxy"; import { createRequestDebugSession, isRequestDebugEnabled, type RequestDebugResponseLog } from "../utils/request-debug"; import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema"; import { notifyRawSseEvent } from "../utils/sse-debug"; @@ -111,7 +112,7 @@ import { promoteResponsesToolUseStopReason, type SequentialCutoffSummaryState, } from "./openai-shared"; -import { transformMessages } from "./transform-messages"; +import { redactSensitiveInObject, transformMessages } from "./transform-messages"; export interface OpenAICodexResponsesOptions extends StreamOptions { reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; @@ -1481,6 +1482,7 @@ async function openCodexWebSocketTransport( websocketState, toWebSocketUrl(requestContext.url), websocketHeaders, + model.provider, requestSetup.requestSignal, ); const eventStream = websocketConnection.streamRequest( @@ -2614,6 +2616,7 @@ export async function prewarmOpenAICodexResponses( state, toWebSocketUrl(url), headers, + model.provider, options?.signal, ); state.prewarmed = true; @@ -3076,11 +3079,13 @@ interface CodexWebSocketRequestTimeouts { interface CodexWebSocketConnectionOptions { onHandshakeHeaders?: (headers: Headers) => void; + proxy?: string; } class CodexWebSocketConnection { #url: string; #headers: Record; + #proxy?: string; #onHandshakeHeaders?: (headers: Headers) => void; #socket: Bun.WebSocket | null = null; #queue: Array | Error | null> = []; @@ -3111,6 +3116,7 @@ class CodexWebSocketConnection { constructor(url: string, headers: Record, options: CodexWebSocketConnectionOptions) { this.#url = url; this.#headers = headers; + this.#proxy = options.proxy; this.#onHandshakeHeaders = options.onHandshakeHeaders; } @@ -3172,7 +3178,7 @@ class CodexWebSocketConnection { this.#connectPromise = promise; const socket = new (WebSocket as unknown as new (url: string, opts: Bun.WebSocketOptions) => Bun.WebSocket)( this.#url, - { headers: this.#headers }, + { headers: this.#headers, proxy: this.#proxy }, ); socket.binaryType = "nodebuffer"; this.#socket = socket; @@ -3663,8 +3669,18 @@ async function getOrCreateCodexWebSocketConnection( state: CodexWebSocketSessionState, url: string, headers: Headers, + provider: string, signal?: AbortSignal, ): Promise { + const targetUrl = new URL(url); + const proxy = shouldBypassProxy(targetUrl) + ? undefined + : (getProxyForProvider(provider) ?? + (targetUrl.protocol === "wss:" + ? Bun.env.HTTPS_PROXY || Bun.env.https_proxy + : Bun.env.HTTP_PROXY || Bun.env.http_proxy) ?? + Bun.env.ALL_PROXY ?? + Bun.env.all_proxy); const headerRecord = headersToRecord(headers); // Join an in-flight handshake instead of tearing it down: closing a // CONNECTING socket rejects the concurrent caller (prewarm racing the first @@ -3707,6 +3723,7 @@ async function getOrCreateCodexWebSocketConnection( onHandshakeHeaders: handshakeHeaders => { updateCodexSessionMetadataFromHeaders(state, handshakeHeaders); }, + proxy, }); await state.connection.connect(signal); return state.connection; @@ -3896,7 +3913,8 @@ function redactHeaders(headers: Headers): Record { return redacted; } -function resolveCodexResponsesUrl(baseUrl: string | undefined): string { +/** Resolve a Codex Responses endpoint exactly as the chat and compaction transports do. */ +export function resolveCodexResponsesUrl(baseUrl: string | undefined): string { const raw = baseUrl && baseUrl.trim().length > 0 ? baseUrl : CODEX_BASE_URL; const normalized = raw.replace(/\/+$/, ""); if (normalized.endsWith("/codex/responses")) return normalized; @@ -3937,13 +3955,14 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex | Array | undefined; if (historyItems) { - for (const item of historyItems) { + const redactedHistoryItems = redactSensitiveInObject(historyItems).result as Array; + for (const item of redactedHistoryItems) { const maybe = item as { type?: string; call_id?: string }; if (maybe.type === "custom_tool_call" && typeof maybe.call_id === "string") { customCallIds.add(maybe.call_id); } } - messages.push(...historyItems); + messages.push(...redactedHistoryItems); msgIndex += 1; continue; } diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index d21e816b4..d7887485f 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -269,6 +269,7 @@ function stripImageDetails(input: unknown[]): void { export interface CodexLiteShapedBody { instructions?: unknown; tools?: unknown; + tool_choice?: unknown; input?: unknown; parallel_tool_calls?: unknown; } @@ -278,9 +279,14 @@ export interface CodexLiteShapedBody { * `build_responses_request` with `use_responses_lite`): strips pinned image * detail, forces parallel tool calling off, moves tools into a leading * `additional_tools` developer item and the base instructions into a - * developer message, then omits top-level `instructions`/`tools`. Shared by - * normal turns and both remote-compaction paths — codex-rs routes - * `/responses/compact` through the same builder. + * developer message, then omits top-level `instructions`/`tools`. Because the + * rewrite removes top-level `tools`, a forced hosted-tool choice (e.g. + * `{ type: "web_search" }`) would leave the backend unable to validate the + * choice against a tools collection and it rejects the request with HTTP 400 + * (#5771). Such choices must fall back to `"auto"`; explicit string constraints + * such as `"none"` and `"required"` remain valid. Shared by normal turns and + * both remote-compaction paths — codex-rs routes `/responses/compact` through + * the same builder. */ export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void { const input = Array.isArray(body.input) ? body.input : []; @@ -297,6 +303,9 @@ export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void { }); } body.input = [...prefix, ...input]; + if (body.tool_choice !== "none" && body.tool_choice !== "required") { + body.tool_choice = "auto"; + } delete body.instructions; delete body.tools; } diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 83cc11666..df46e01f2 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -42,7 +42,13 @@ import { import { OpenAIHttpError, postOpenAIStream } from "../utils/openai-http"; import { notifyProviderResponse } from "../utils/provider-response"; import { callWithCopilotModelRetry } from "../utils/retry"; -import { adaptSchemaForStrict, NO_STRICT, normalizeSchemaForMoonshot, toolWireSchema } from "../utils/schema"; +import { + adaptSchemaForStrict, + NO_STRICT, + normalizeSchemaForMoonshot, + sanitizeSchemaForGrammar, + toolWireSchema, +} from "../utils/schema"; import { type HealedToolCall, StreamMarkupHealing, @@ -76,6 +82,7 @@ import { applyOpenAIExtraBody, applyOpenAIGatewayRouting, applyOpenAIServiceTier, + applyOpenRouterReportedCost, applyWireModelIdTransform, calculateOpenAIUsageAccounting, clearOpenAIStrictToolsState, @@ -93,9 +100,9 @@ import { type OpenAIStrictToolsState, parseAzureDeploymentNameMap, resolveOpenAICompatPolicy, + resolveOpenAICompletionsOutputClamp, resolveOpenAIOutputTokenParam, resolveOpenAIRequestSetup, - resolveZaiReasoningOutputClamp, shouldRetryWithoutStrictTools, } from "./openai-shared"; import { transformMessages } from "./transform-messages"; @@ -670,7 +677,7 @@ const streamOpenAICompletionsOnce = ( } activeReasoningEffortFallbackKey = reasoningEffortFallbackKey; activeRequestParams = params; - options?.onPayload?.(params); + options?.onPayload?.(params, model); rawRequestDump = { provider: model.provider, api: output.api, @@ -1430,6 +1437,20 @@ function dropOpenRouterKimiForcedToolReasoning( } } +function hasActiveNativeKimiK3Reasoning( + model: Model<"openai-completions">, + options: OpenAICompletionsOptions | undefined, +): boolean { + if (model.provider !== "kimi-code" || model.id.toLowerCase() !== "k3" || !model.reasoning) return false; + if (options?.reasoning === undefined || options.disableReasoning) return false; + try { + const url = new URL(model.baseUrl); + return url.hostname === "api.kimi.com" && (url.pathname === "/coding" || url.pathname.startsWith("/coding/")); + } catch { + return false; + } +} + function buildParams( model: Model<"openai-completions">, context: Context, @@ -1460,31 +1481,35 @@ function buildParams( params.store = false; } - if (options?.temperature !== undefined) { - params.temperature = options.temperature; - } - if (options?.topP !== undefined) { - params.top_p = options.topP; - } - if (options?.topK !== undefined) { - params.top_k = options.topK; - } - if (options?.minP !== undefined) { - params.min_p = options.minP; - } - if (options?.presencePenalty !== undefined) { - params.presence_penalty = options.presencePenalty; - } - if (options?.repetitionPenalty !== undefined) { - params.repetition_penalty = options.repetitionPenalty; + // OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit + // sampling params with a 400 on every serving host (#5606). + if (initialCompat.supportsSamplingParams) { + if (options?.temperature !== undefined) { + params.temperature = options.temperature; + } + if (options?.topP !== undefined) { + params.top_p = options.topP; + } + if (options?.topK !== undefined) { + params.top_k = options.topK; + } + if (options?.minP !== undefined) { + params.min_p = options.minP; + } + if (options?.presencePenalty !== undefined) { + params.presence_penalty = options.presencePenalty; + } + if (options?.repetitionPenalty !== undefined) { + params.repetition_penalty = options.repetitionPenalty; + } + if (options?.frequencyPenalty !== undefined) { + params.frequency_penalty = options.frequencyPenalty; + } } if (options?.stopSequences?.length) { const seqs = options.stopSequences; params.stop = seqs.length === 1 ? seqs[0] : seqs.slice(0, 4); } - if (options?.frequencyPenalty !== undefined) { - params.frequency_penalty = options.frequencyPenalty; - } applyOpenAIServiceTier(params, options?.serviceTier, model); if (context.tools?.length) { @@ -1513,6 +1538,20 @@ function buildParams( ) { params.tool_choice = "required"; } + const forcedToolName = + typeof params.tool_choice === "object" && params.tool_choice !== null && "function" in params.tool_choice + ? params.tool_choice.function.name + : undefined; + if ( + forcedToolName !== undefined && + Array.isArray(params.tools) && + params.tools.some(tool => tool.type === "function" && tool.function.name === forcedToolName) && + hasActiveNativeKimiK3Reasoning(model, options) + ) { + // Native K3 reasoning is incompatible with selecting a specific function. + // Preserve the hard tool-use contract while letting K3 choose among tools. + params.tool_choice = "required"; + } if (isForcedToolChoice(params.tool_choice) && !initialCompat.supportsForcedToolChoice) { // Some thinking-required OpenAI-compatible models reject forced // `tool_choice` while still accepting tools with the default auto @@ -1532,10 +1571,6 @@ function buildParams( delete params.tool_choice; } - const forcedToolName = - typeof params.tool_choice === "object" && params.tool_choice !== null && "function" in params.tool_choice - ? params.tool_choice.function.name - : undefined; if ( forcedToolName !== undefined && (!Array.isArray(params.tools) || @@ -1566,7 +1601,7 @@ function buildParams( omitMaxOutputTokens: model.omitMaxOutputTokens ?? false, isOpenRouterHost: compat.isOpenRouterHost, alwaysSendMaxTokens: compat.alwaysSendMaxTokens, - providerOutputClamp: resolveZaiReasoningOutputClamp(model, compat), + providerOutputClamp: resolveOpenAICompletionsOutputClamp(model, compat), }); if (outputToken) { if (outputToken.field === "max_tokens") { @@ -1629,6 +1664,7 @@ export function parseChunkUsage( ...(premiumRequests !== undefined ? { premiumRequests } : {}), }; calculateCost(model, usage); + applyOpenRouterReportedCost(model, usage, rawUsage); return usage; } @@ -2199,10 +2235,16 @@ function convertTools( description: tool.description || "", // Moonshot/Kimi native hosts validate against the stricter MFJS subset // (const→enum, typed enums, no validators) and 400 otherwise. + // Grammar-constrained local backends (llama.cpp, LM Studio, vLLM) + // build a GBNF grammar from the schema and 400 with + // `Unrecognized schema: true` on the bare boolean subschema + // `toolWireSchema` emits for open fields (issue #5914). parameters: compat.toolSchemaFlavor === "moonshot-mfjs" ? (normalizeSchemaForMoonshot(wireParameters) as Record) - : wireParameters, + : compat.toolSchemaFlavor === "grammar" + ? sanitizeSchemaForGrammar(wireParameters) + : wireParameters, // Only include strict if provider supports it. Some reject unknown fields. ...(includeStrict ? { strict: true } : includeExplicitFalse ? { strict: false } : {}), }, diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 2be20dd81..96c33beb6 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -1,3 +1,4 @@ +import { scheduler } from "node:timers/promises"; import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; import { $flag, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; @@ -38,6 +39,7 @@ import { adaptSchemaForStrict, findStrictToolSchemaViolation, NO_STRICT, + normalizeSchemaForMoonshot, sanitizeSchemaForOpenAIResponses, toolWireSchema, } from "../utils/schema"; @@ -154,6 +156,33 @@ const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE = "OpenAI responses stream timed out while waiting for the first event"; /** Consecutive stale-previous-response failures before chaining is disabled for the session. */ const OPENAI_RESPONSES_CHAIN_STALE_FAILURE_LIMIT = 3; +const OPENAI_RESPONSES_MAX_TRANSIENT_STREAM_RETRIES = 1; +const OPENAI_RESPONSES_TRANSIENT_STREAM_RETRY_DELAY_MS = 500; + +function isOpenAIResponsesReplayUnsafeEvent(event: ResponseStreamEvent): boolean { + switch (event.type) { + case "response.output_text.delta": + case "response.refusal.delta": + case "response.reasoning_summary_text.delta": + case "response.reasoning_text.delta": + case "response.function_call_arguments.delta": + case "response.custom_tool_call_input.delta": + return typeof event.delta === "string" && event.delta.length > 0; + case "response.reasoning_summary_part.done": + return true; + case "response.output_item.done": + return true; + default: + return false; + } +} + +function isRetryableOpenAIResponsesStreamFailure(error: unknown): boolean { + return ( + AIError.isTransientStreamParseError(error) || + (error instanceof AIError.ProviderResponseError && error.kind === "incomplete-stream") + ); +} interface OpenAIResponsesProviderSessionState extends ProviderSessionState, @@ -452,7 +481,7 @@ const streamOpenAIResponsesOnce = ( return payload; }; chained = { ...chained, params: await applyPayloadReplacement(chained.params) }; - rawRequestDump = { + const activeRawRequestDump: RawHttpRequestDump = { provider: model.provider, api: output.api, model: model.id, @@ -460,6 +489,7 @@ const streamOpenAIResponsesOnce = ( url: requestUrl, body: chained.params, }; + rawRequestDump = activeRawRequestDump; const openResponsesStream = (requestParams: OpenAIResponsesSamplingParams) => { activeReasoningEffortFallbackKey = createOpenAIReasoningEffortFallbackKey( "responses", @@ -507,183 +537,257 @@ const streamOpenAIResponsesOnce = ( { provider: model.provider, signal: requestSignal }, ); }; - let openaiStream: AsyncIterable; let strictRetryAvailable = true; let activeStrictToolsApplied = builtParams.strictToolsApplied; let forceDisableStrictTools = false; - while (true) { - try { - openaiStream = await openResponsesStream(chained.params); - if (pendingReasoningEffortFallback) { - rememberOpenAIReasoningEffortFallback( - providerSessionState, - pendingReasoningEffortFallback.key, - pendingReasoningEffortFallback.fallback, - ); - pendingReasoningEffortFallback = undefined; - } - break; - } catch (error) { - const capturedErrorResponse = error instanceof OpenAIHttpError ? error.captured : undefined; - const reasoningEffortFallback = - activeReasoningEffortFallbackKey && activeRequestParams && !requestSignal.aborted - ? resolveOpenAIReasoningEffortFallback(error, capturedErrorResponse, activeRequestParams, { - explicitDisable: options?.disableReasoning === true && options.reasoning === undefined, - }) - : undefined; - if (reasoningEffortFallback !== undefined && activeReasoningEffortFallbackKey) { - const retryMarker = `${activeReasoningEffortFallbackKey}:${String(reasoningEffortFallback)}`; - if (attemptedReasoningEffortFallbacks.has(retryMarker)) throw error; - attemptedReasoningEffortFallbacks.add(retryMarker); - requestReasoningEffortFallbacks.set(activeReasoningEffortFallbackKey, reasoningEffortFallback); - applyOpenAIReasoningEffortFallback(chained.params, reasoningEffortFallback); - applyOpenAIReasoningEffortFallback(activeParams, reasoningEffortFallback); - rawRequestDump.body = chained.params; - pendingReasoningEffortFallback = { - key: activeReasoningEffortFallbackKey, - fallback: reasoningEffortFallback, - }; - continue; - } - const compiledGrammarTooLarge = - isOpenRouterAnthropicModel(model) && - isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse); - const canRetryWithoutStrictTools = - strictRetryAvailable && - !requestSignal.aborted && - (compiledGrammarTooLarge || - shouldRetryWithoutStrictTools( - error, - capturedErrorResponse, - activeStrictToolsApplied, - context.tools, - )); - if (canRetryWithoutStrictTools) { - strictRetryAvailable = false; - forceDisableStrictTools = true; - disableStrictToolsForScope(providerSessionState, strictToolsScope); - const fallbackBuilt = buildParams( + const openResponsesStreamWithFallbacks = async (): Promise> => { + let openaiStream: AsyncIterable; + while (true) { + try { + openaiStream = await openResponsesStream(chained.params); + if (pendingReasoningEffortFallback) { + rememberOpenAIReasoningEffortFallback( + providerSessionState, + pendingReasoningEffortFallback.key, + pendingReasoningEffortFallback.fallback, + ); + pendingReasoningEffortFallback = undefined; + } + break; + } catch (error) { + const capturedErrorResponse = error instanceof OpenAIHttpError ? error.captured : undefined; + const reasoningEffortFallback = + activeReasoningEffortFallbackKey && activeRequestParams && !requestSignal.aborted + ? resolveOpenAIReasoningEffortFallback(error, capturedErrorResponse, activeRequestParams, { + explicitDisable: options?.disableReasoning === true && options.reasoning === undefined, + }) + : undefined; + if (reasoningEffortFallback !== undefined && activeReasoningEffortFallbackKey) { + const retryMarker = `${activeReasoningEffortFallbackKey}:${String(reasoningEffortFallback)}`; + if (attemptedReasoningEffortFallbacks.has(retryMarker)) throw error; + attemptedReasoningEffortFallbacks.add(retryMarker); + requestReasoningEffortFallbacks.set(activeReasoningEffortFallbackKey, reasoningEffortFallback); + applyOpenAIReasoningEffortFallback(chained.params, reasoningEffortFallback); + applyOpenAIReasoningEffortFallback(activeParams, reasoningEffortFallback); + activeRawRequestDump.body = chained.params; + pendingReasoningEffortFallback = { + key: activeReasoningEffortFallbackKey, + fallback: reasoningEffortFallback, + }; + continue; + } + const compiledGrammarTooLarge = + isOpenRouterAnthropicModel(model) && + isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse); + const canRetryWithoutStrictTools = + strictRetryAvailable && + !requestSignal.aborted && + (compiledGrammarTooLarge || + shouldRetryWithoutStrictTools( + error, + capturedErrorResponse, + activeStrictToolsApplied, + context.tools, + )); + if (canRetryWithoutStrictTools) { + strictRetryAvailable = false; + forceDisableStrictTools = true; + disableStrictToolsForScope(providerSessionState, strictToolsScope); + const fallbackBuilt = buildParams( + model, + context, + options, + providerSessionState, + strictToolsScope, + true, + ); + const fallbackParams = fallbackBuilt.params; + if (chainState && !chainState.disabled) fallbackParams.store = true; + let fallbackChained: OpenAIResponsesChainedParams = + chainState && !chainState.disabled + ? buildOpenAIResponsesChainedParams(fallbackParams, chainState) + : { params: fallbackParams }; + sentPreviousResponseId = fallbackChained.previousResponseId; + fallbackChained = { + ...fallbackChained, + params: await applyPayloadReplacement(fallbackChained.params), + }; + chained = fallbackChained; + activeRawRequestDump.body = chained.params; + activeParams = fallbackParams; + activeStrictToolsApplied = fallbackBuilt.strictToolsApplied; + continue; + } + if (!chainState || !sentPreviousResponseId || requestSignal.aborted) { + throw error; + } + const zdrRejection = + error instanceof Error && + /previous[ _]?response/i.test(error.message) && + /zero[ _-]?data[ _-]?retention/i.test(error.message); + const isPromptBlocked = + error instanceof Error && + ((error as { code?: string }).code === "invalid_prompt" || + /invalid_prompt|Request blocked/i.test(error.message)); + if (!zdrRejection && !isPromptBlocked && !isOpenAIResponsesStalePreviousResponseError(error)) { + throw error; + } + // Server rejected the chain baseline: reset, count the failure (or + // disable categorically on ZDR), and retry once with the full + // transcript. Structurally cannot loop — the retry carries no + // previous_response_id. + if (zdrRejection) { + markOpenAIResponsesChainZeroDataRetention(chainState, error); + // ZDR orgs cannot store responses; the retry uses `store: false`. + } else { + registerOpenAIResponsesChainStaleFailure(chainState, error); + } + sentPreviousResponseId = undefined; + const currentBuilt = buildParams( model, context, options, providerSessionState, strictToolsScope, - true, + forceDisableStrictTools, ); - const fallbackParams = fallbackBuilt.params; - if (chainState && !chainState.disabled) fallbackParams.store = true; - let fallbackChained: OpenAIResponsesChainedParams = - chainState && !chainState.disabled - ? buildOpenAIResponsesChainedParams(fallbackParams, chainState) - : { params: fallbackParams }; - sentPreviousResponseId = fallbackChained.previousResponseId; - fallbackChained = { - ...fallbackChained, - params: await applyPayloadReplacement(fallbackChained.params), - }; - chained = fallbackChained; - rawRequestDump.body = chained.params; - activeParams = fallbackParams; - activeStrictToolsApplied = fallbackBuilt.strictToolsApplied; - continue; + const currentParams = currentBuilt.params; + // Only ZDR forces `store: false` (the org never persists responses). A + // non-ZDR stale baseline is transient, so keep storing: the full-context + // retry must be chainable next turn, and the consecutive stale-failure + // breaker only trips when each retry stores and the next turn re-chains. + currentParams.store = !zdrRejection; + const retryParams = await applyPayloadReplacement(currentParams); + chained = { params: retryParams }; + activeRawRequestDump.body = retryParams; + activeParams = currentParams; + activeStrictToolsApplied = currentBuilt.strictToolsApplied; } - if (!chainState || !sentPreviousResponseId || requestSignal.aborted) { - throw error; - } - const zdrRejection = - error instanceof Error && - /previous[ _]?response/i.test(error.message) && - /zero[ _-]?data[ _-]?retention/i.test(error.message); - if (!zdrRejection && !isOpenAIResponsesStalePreviousResponseError(error)) { - throw error; - } - // Server rejected the chain baseline: reset, count the failure (or - // disable categorically on ZDR), and retry once with the full - // transcript. Structurally cannot loop — the retry carries no - // previous_response_id. - if (zdrRejection) { - markOpenAIResponsesChainZeroDataRetention(chainState, error); - // ZDR orgs cannot store responses; the retry uses `store: false`. - } else { - registerOpenAIResponsesChainStaleFailure(chainState, error); - } - sentPreviousResponseId = undefined; - const currentBuilt = buildParams( - model, - context, - options, - providerSessionState, - strictToolsScope, - forceDisableStrictTools, - ); - const currentParams = currentBuilt.params; - // Only ZDR forces `store: false` (the org never persists responses). A - // non-ZDR stale baseline is transient, so keep storing: the full-context - // retry must be chainable next turn, and the consecutive stale-failure - // breaker only trips when each retry stores and the next turn re-chains. - currentParams.store = !zdrRejection; - const retryParams = await applyPayloadReplacement(currentParams); - chained = { params: retryParams }; - rawRequestDump.body = retryParams; - activeParams = currentParams; - activeStrictToolsApplied = currentBuilt.strictToolsApplied; } - } + return openaiStream; + }; + let openaiStream = await openResponsesStreamWithFallbacks(); if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal; stream.push({ type: "start", partial: output }); const nativeOutputItems: Array> = []; - let sawTerminalResponseEvent = false; - const timedOpenaiStream = iterateWithIdleTimeout(openaiStream, { - idleTimeoutMs, - firstItemTimeoutMs: firstEventTimeoutMs, - firstItemErrorMessage: OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE, - errorMessage: "OpenAI responses stream stalled while waiting for the next event", - onFirstItemTimeout: () => abortTracker.abortLocally(firstEventTimeoutAbortError), - onIdle: () => requestAbortController.abort(), - abortSignal: options?.signal, - isProgressItem: isOpenAIResponsesProgressEvent, - }); - await processResponsesStream(timedOpenaiStream, output, stream, model, { - onFirstToken: () => { - if (!firstTokenTime) firstTokenTime = performance.now(); - }, - onOutputItemDone: item => { - // `processResponsesStream` hands over a private clone already; no - // second deep copy needed (reasoning items carry multi-KB blobs). - nativeOutputItems.push(item as unknown as Record); - }, - onCompleted: () => { - sawTerminalResponseEvent = true; - }, - requestServiceTier: options?.serviceTier, - }); - - const localAbortReason = abortTracker.getLocalAbortReason(); - if (localAbortReason) { - throw localAbortReason; - } - if (abortTracker.wasCallerAbort()) { - throw new AIError.AbortError(); - } - - // Detect premature stream closure: the HTTP stream ended without the - // provider sending a recognized terminal response event. - // Custom/proxy providers may drop the connection mid-stream; without - // this guard the incomplete output is silently surfaced as a successful - // "stop". - if (!sawTerminalResponseEvent) { - throw new AIError.ProviderResponseError( - "OpenAI responses stream closed before a terminal response event was received", - { provider: model.provider, kind: "incomplete-stream" }, - ); - } - - if (output.stopReason === "aborted" || output.stopReason === "error") { - throw new AIError.ProviderResponseError(output.errorMessage ?? "An unknown error occurred", { - provider: model.provider, - kind: "runtime", + let transientStreamRetryAttempt = 0; + while (true) { + let sawReplayUnsafeOutput = false; + let sawTerminalResponseEvent = false; + const attemptStream = new AssistantMessageEventStream(); + let forwardAttemptLive = false; + const forwardAttemptEvents = () => { + for (const event of attemptStream.queue) stream.push(event); + attemptStream.queue.length = 0; + }; + nativeOutputItems.length = 0; + const timedOpenaiStream = iterateWithIdleTimeout(openaiStream, { + idleTimeoutMs, + firstItemTimeoutMs: firstEventTimeoutMs, + firstItemErrorMessage: OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE, + errorMessage: "OpenAI responses stream stalled while waiting for the next event", + onFirstItemTimeout: () => abortTracker.abortLocally(firstEventTimeoutAbortError), + onIdle: () => requestAbortController.abort(), + abortSignal: options?.signal, + isProgressItem: isOpenAIResponsesProgressEvent, }); + const observedOpenaiStream = (async function* (): AsyncGenerator { + for await (const event of timedOpenaiStream) { + if (isOpenAIResponsesReplayUnsafeEvent(event)) { + sawReplayUnsafeOutput = true; + if (!forwardAttemptLive) { + forwardAttemptEvents(); + forwardAttemptLive = true; + } + } + yield event; + if (forwardAttemptLive) forwardAttemptEvents(); + } + })(); + + try { + await processResponsesStream(observedOpenaiStream, output, attemptStream, model, { + onFirstToken: () => { + if (!firstTokenTime) firstTokenTime = performance.now(); + }, + onOutputItemDone: item => { + // `processResponsesStream` hands over a private clone already; no + // second deep copy needed (reasoning items carry multi-KB blobs). + nativeOutputItems.push(item as unknown as Record); + }, + onCompleted: () => { + sawTerminalResponseEvent = true; + }, + requestServiceTier: options?.serviceTier, + }); + + const localAbortReason = abortTracker.getLocalAbortReason(); + if (localAbortReason) throw localAbortReason; + if (abortTracker.wasCallerAbort()) throw new AIError.AbortError(); + + // Detect premature stream closure: the HTTP stream ended without the + // provider sending a recognized terminal response event. + if (!sawTerminalResponseEvent) { + throw new AIError.ProviderResponseError( + "OpenAI responses stream closed before a terminal response event was received", + { provider: model.provider, kind: "incomplete-stream" }, + ); + } + + if (output.stopReason === "aborted" || output.stopReason === "error") { + throw new AIError.ProviderResponseError(output.errorMessage ?? "An unknown error occurred", { + provider: model.provider, + kind: "runtime", + }); + } + forwardAttemptEvents(); + break; + } catch (error) { + const streamFailure = abortTracker.getLocalAbortReason() ?? error; + const canRetry = + !sawReplayUnsafeOutput && + !requestSignal.aborted && + !abortTracker.wasCallerAbort() && + transientStreamRetryAttempt < OPENAI_RESPONSES_MAX_TRANSIENT_STREAM_RETRIES && + isRetryableOpenAIResponsesStreamFailure(streamFailure); + if (!canRetry) { + forwardAttemptEvents(); + throw streamFailure; + } + + transientStreamRetryAttempt++; + logger.debug("OpenAI responses stream ended before replay-unsafe output; retrying", { + provider: model.provider, + model: model.id, + attempt: transientStreamRetryAttempt, + error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure), + }); + const retryOutput = createInitialResponsesAssistantMessage(model.api, model.provider, model.id); + output.content.length = 0; + output.responseId = undefined; + output.upstreamProvider = undefined; + output.errorMessage = undefined; + output.errorStatus = undefined; + output.errorId = undefined; + output.stopDetails = undefined; + output.providerPayload = undefined; + output.usage = retryOutput.usage; + if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal; + output.stopReason = "stop"; + output.duration = undefined; + output.ttft = undefined; + firstTokenTime = undefined; + nativeOutputItems.length = 0; + + if (options?.providerRetryWait) { + await options.providerRetryWait(OPENAI_RESPONSES_TRANSIENT_STREAM_RETRY_DELAY_MS, options.signal); + } else { + await scheduler.wait(OPENAI_RESPONSES_TRANSIENT_STREAM_RETRY_DELAY_MS, { signal: options?.signal }); + } + if (abortTracker.wasCallerAbort()) throw new AIError.AbortError(); + openaiStream = await openResponsesStreamWithFallbacks(); + } } output.providerPayload = createOpenAIResponsesHistoryPayload(model.provider, nativeOutputItems); @@ -995,7 +1099,15 @@ export function convertTools( } const strict = !NO_STRICT && strictMode && tool.strict !== false; const baseParameters = toolWireSchema(tool); - const responseParameters = sanitizeSchemaForOpenAIResponses(baseParameters); + // MFJS must run AFTER the Responses sanitizer: the sanitizer normalizes + // `{}` → `true` (issue #1179), and Moonshot's validator rejects boolean + // subschemas ("property schema … must be an object"), so the Moonshot + // pass re-coerces them last. + const sanitized = sanitizeSchemaForOpenAIResponses(baseParameters); + const responseParameters = + model.compat.toolSchemaFlavor === "moonshot-mfjs" + ? (normalizeSchemaForMoonshot(sanitized) as Record) + : sanitized; const { schema: parameters, strict: effectiveStrict } = adaptSchemaForStrict(responseParameters, strict); // Quarantine a tool whose emitted schema carries a provider-rejecting // enum/const-vs-type contradiction: dropping just that tool keeps the rest diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 64189c5e6..32112b8f3 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1,6 +1,6 @@ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; -import { isGlm52ReasoningEffortModelId } from "@oh-my-pi/pi-catalog/identity"; +import { isGlm52ReasoningEffortModelId, isKimiK3ModelId } from "@oh-my-pi/pi-catalog/identity"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import type { @@ -25,6 +25,7 @@ import { classifyJsonPrefix, extractHttpStatusFromError, logger, + parseImageMetadata, parseStreamingJson, parseStreamingJsonThrottled, stringifyJson, @@ -100,6 +101,16 @@ import type { import { transformMessages } from "./transform-messages"; import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard"; +/** + * Keyless-provider sentinel. Custom providers configured with `auth: none` + * (models.yml) have no credential, so the coding-agent resolves their API key + * to this literal instead of a real secret. Providers must treat it as "no + * credential" and suppress any credential-bearing header (e.g. `Authorization: + * Bearer …`) rather than forwarding the sentinel on the wire. See #6188; the + * google-vertex and amazon-bedrock transports apply the same guard inline. + */ +export const NO_AUTH_SENTINEL = "N/A"; + export interface OpenAIModelIdentity { provider: string; id: string; @@ -278,7 +289,15 @@ export function resolveOpenAIRequestSetup( baseUrl = baseUrl ?? ($env.OPENAI_BASE_URL?.trim() || options.defaultBaseUrl); } const requestHeaders = { ...headers }; - headers.Authorization ??= `Bearer ${apiKey}`; + // A keyless provider (`auth: none` in models.yml) resolves to the `N/A` + // sentinel rather than a real key. Injecting `Authorization: Bearer N/A` + // breaks custom endpoints that authenticate via their own headers (e.g. + // `headers.x-api-key`) and reject the bogus bearer — mirror the sentinel + // guards in google-vertex / amazon-bedrock and send no Authorization here + // (#6188). A caller-supplied Authorization in `model.headers` still wins. + if (apiKey !== NO_AUTH_SENTINEL) { + headers.Authorization ??= `Bearer ${apiKey}`; + } return { copilotPremiumRequests, baseUrl, headers, query, requestHeaders }; } @@ -338,6 +357,29 @@ export function applyOpenAIResponsesServiceTierCost( usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite; } +/** Reconcile token-price estimates with OpenRouter's authoritative account charge. */ +export function applyOpenRouterReportedCost(model: Pick, usage: Usage, rawUsage: unknown): void { + if (model.provider !== "openrouter" || typeof rawUsage !== "object" || rawUsage === null) return; + const reportedCost = Reflect.get(rawUsage, "cost"); + if (typeof reportedCost !== "number" || !Number.isFinite(reportedCost) || reportedCost < 0) return; + + const estimatedCost = usage.cost.total; + if (Number.isFinite(estimatedCost) && estimatedCost > 0) { + const scale = reportedCost / estimatedCost; + usage.cost.input *= scale; + usage.cost.output *= scale; + usage.cost.cacheRead *= scale; + usage.cost.cacheWrite *= scale; + } else { + // Keep legacy component-only aggregators additive when catalog pricing is unavailable. + usage.cost.input = reportedCost; + usage.cost.output = 0; + usage.cost.cacheRead = 0; + usage.cost.cacheWrite = 0; + } + usage.cost.total = reportedCost; +} + export interface OpenAIUsageAccountingInput { promptTokens: number; outputTokens: number; @@ -630,7 +672,7 @@ export type OpenAICompletionsParams = Omit, compat: } /** - * Output-token clamp for the Z.AI/GLM-5.2 reasoning dialect: these hosts accept - * the full model window on reasoning turns, so clamp to the model cap. Returns - * `undefined` for every other model, leaving {@link resolveOpenAIOutputTokenParam} - * on its default `OPENAI_MAX_OUTPUT_TOKENS` clamp. + * Provider-specific Chat Completions output clamp. + * + * Most OpenAI-compatible endpoints retain the conservative 64k ceiling from + * {@link resolveOpenAIOutputTokenParam}. Z.AI/GLM-5.2 reasoning and native + * Moonshot K3 explicitly accept their full advertised model caps, so those + * routes clamp to `model.maxTokens` instead. */ -export function resolveZaiReasoningOutputClamp( +export function resolveOpenAICompletionsOutputClamp( model: Model<"openai-completions">, compat: ResolvedOpenAICompat, ): number | undefined { - return isZaiReasoningEffortDialect(model, compat) ? (model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS) : undefined; + if (isZaiReasoningEffortDialect(model, compat)) { + return model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS; + } + if (model.provider === "moonshot" && isKimiK3ModelId(model.id)) { + return model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS; + } + return undefined; } /** @@ -1702,6 +1757,21 @@ export function convertResponsesAssistantMessage( return outputItems; } +const syntheticToolImageMessages = new WeakSet(); + +function insertResponsesToolOutput(messages: ResponseInput, output: ResponseInput[number]): void { + let index = messages.length; + while (index > 0) { + const previous = messages[index - 1]; + if (typeof previous !== "object" || previous === null || !syntheticToolImageMessages.has(previous)) { + break; + } + index -= 1; + } + messages.splice(index, 0, output); +} + +/** Appends one tool result while keeping consecutive outputs ahead of its synthetic image messages. */ export function appendResponsesToolResultMessages( messages: ResponseInput, toolResult: ToolResultMessage, @@ -1748,13 +1818,13 @@ export function appendResponsesToolResultMessages( return; } if (supportsCustomToolCalls && customCallIds?.has(normalized.callId)) { - messages.push({ + insertResponsesToolOutput(messages, { type: "custom_tool_call_output", call_id: normalized.callId, output, } as ResponseInput[number]); } else { - messages.push({ + insertResponsesToolOutput(messages, { type: "function_call_output", call_id: normalized.callId, output, @@ -1777,7 +1847,9 @@ export function appendResponsesToolResultMessages( } satisfies ResponseInputImage); } } - messages.push({ role: "user", content: contentParts }); + const imageMessage = { role: "user", content: contentParts } satisfies ResponseInput[number]; + syntheticToolImageMessages.add(imageMessage); + messages.push(imageMessage); } /** @@ -2429,6 +2501,7 @@ export async function processResponsesStream( const entry = lookupOpenToolCallAlias(event, "custom_tool_call"); if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") { finalizeCustomToolCallInputDone(entry.block, event.input); + entry.block[kStreamingArgumentsDone] = true; } } else if (event.type === "response.output_item.done") { const item = structuredCloneJSON(event.item); @@ -2532,15 +2605,33 @@ export async function processResponsesStream( } closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id)); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); + } else if (item.type === "image_generation_call" && item.status === "completed" && item.result) { + const image: ImageContent = { + type: "image", + data: item.result, + mimeType: parseImageMetadata(Buffer.from(item.result, "base64"))?.mimeType ?? "image/png", + }; + output.content.push(image); + stream.push({ + type: "image_end", + contentIndex: output.content.length - 1, + content: image, + partial: output, + }); } } else if (terminalEvent) { const response = terminalEvent.response; + const shouldPromoteIncompleteToolUse = + response?.status === "incomplete" && + response.incomplete_details?.reason === "max_output_tokens" && + hasExecutableIncompleteResponsesToolCalls(output); finalizePendingResponsesToolCalls(output); if (response?.id) { output.responseId = response.id; } populateResponsesUsageFromResponse(output, response?.usage); calculateCost(model, output.usage); + applyOpenRouterReportedCost(model, output.usage, response?.usage); applyOpenAIResponsesServiceTierCost( model, output.usage, @@ -2570,7 +2661,11 @@ export async function processResponsesStream( kind: "content-blocked", }); } - promoteResponsesToolUseStopReason(output, (response as { end_turn?: boolean } | undefined)?.end_turn); + promoteResponsesToolUseStopReason( + output, + (response as { end_turn?: boolean } | undefined)?.end_turn, + shouldPromoteIncompleteToolUse, + ); options?.onCompleted?.(); // `response.completed`/`response.incomplete`/`response.done` is the last event of a // Responses stream. Stop pulling instead of waiting for the server to @@ -2626,6 +2721,28 @@ export function mapOpenAIResponsesStopReason(status: ResponseStatus | undefined) } } +function hasExecutableIncompleteResponsesToolCalls(output: AssistantMessage): boolean { + let hasToolCall = false; + for (const block of output.content) { + if (block.type !== "toolCall") continue; + hasToolCall = true; + const pending = block as ToolCall & { + [kStreamingPartialJson]?: string; + [kStreamingArgumentsDone]?: boolean; + }; + const rawArguments = pending[kStreamingPartialJson]; + // `output_item.done` is not positive completion proof: our Responses + // compatibility encoder force-closes still-open calls before forwarding an + // upstream `length` stop. Only an explicit arguments/input-done event sets + // this marker; an open ordinary call can instead prove completion with its + // retained strict-complete JSON. + if (pending[kStreamingArgumentsDone]) continue; + if (pending.customWireName !== undefined || rawArguments === undefined) return false; + if (classifyJsonPrefix(rawArguments) !== "complete") return false; + } + return hasToolCall; +} + /** * Finalize any streamed toolCall block whose `output_item.done` never arrived * (lossy proxy, or a terminal event that raced the per-item done): parse the @@ -2659,8 +2776,15 @@ export function finalizePendingResponsesToolCalls(output: AssistantMessage): voi * re-samples instead of ending. Callers set `output.stopReason` from the wire * status first via {@link mapOpenAIResponsesStopReason}. */ -export function promoteResponsesToolUseStopReason(output: AssistantMessage, endTurn: boolean | undefined): void { - if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") { +export function promoteResponsesToolUseStopReason( + output: AssistantMessage, + endTurn: boolean | undefined, + promoteIncompleteToolUse = false, +): void { + if ( + output.content.some(block => block.type === "toolCall") && + (output.stopReason === "stop" || (promoteIncompleteToolUse && output.stopReason === "length")) + ) { output.stopReason = "toolUse"; } if (endTurn === false && output.stopReason === "stop") { @@ -2717,7 +2841,9 @@ type CommonSamplingOptions = Pick< export function applyCommonResponsesSamplingParams

( params: P, options: CommonSamplingOptions | undefined, - model: Pick, + model: Pick & { + compat: Pick; + }, ): void { if (options?.maxTokens && !model.omitMaxOutputTokens) { params.max_output_tokens = Math.min( @@ -2726,12 +2852,16 @@ export function applyCommonResponsesSamplingParams

( * - Preserves tool call structure (unlike converting to text summaries) * - Injects synthetic "aborted" tool results */ +const SENSITIVE_TOKEN_RE = + /(? pattern.test(secret)).length >= 2; +} + +export function redactSensitiveCredentials(text: string): string { + return text.replace(SENSITIVE_TOKEN_RE, match => { + if (!hasPlausibleCredentialEntropy(match)) return match; + const lower = match.toLowerCase(); + if (lower.startsWith("gh")) { + return "[github_token_redacted]"; + } + if (lower.startsWith("gl")) { + return "[gitlab_token_redacted]"; + } + if (lower.startsWith("sk-ant-")) { + return "[anthropic_token_redacted]"; + } + if (lower.startsWith("sk")) { + return "[openai_token_redacted]"; + } + return "[token_redacted]"; + }); +} + +export function redactSensitiveInObject(val: unknown): { result: unknown; changed: boolean } { + if (typeof val === "string") { + const redacted = redactSensitiveCredentials(val); + return { result: redacted, changed: redacted !== val }; + } + if (Array.isArray(val)) { + let changed = false; + const result = val.map(item => { + const res = redactSensitiveInObject(item); + if (res.changed) changed = true; + return res.result; + }); + return { result, changed }; + } + if (val !== null && typeof val === "object") { + let changed = false; + const res: Record = {}; + for (const [k, v] of Object.entries(val)) { + const sub = redactSensitiveInObject(v); + if (sub.changed) changed = true; + res[k] = sub.result; + } + return { result: res, changed }; + } + return { result: val, changed: false }; +} + +function redactSensitiveCredentialsInMessages(messages: Message[]): Message[] { + return messages.map((msg): Message => { + if (msg.role === "user" || msg.role === "developer") { + const userMsg = msg as UserMessage | DeveloperMessage; + if (typeof userMsg.content === "string") { + const redacted = redactSensitiveCredentials(userMsg.content); + if (redacted === userMsg.content) return msg; + return { ...userMsg, content: redacted } as Message; + } + const contentArray = userMsg.content; + let changed = false; + const content = contentArray.map((block): UserMessage["content"][number] => { + if (block.type === "text") { + const redacted = redactSensitiveCredentials(block.text); + if (redacted !== block.text) { + changed = true; + return { ...block, text: redacted }; + } + } + return block; + }); + return (changed ? { ...userMsg, content } : userMsg) as Message; + } + + if (msg.role === "toolResult") { + const toolResultMsg = msg as ToolResultMessage; + let changed = false; + const content = toolResultMsg.content.map((block): ToolResultMessage["content"][number] => { + if (block.type === "text") { + const redacted = redactSensitiveCredentials(block.text); + if (redacted !== block.text) { + changed = true; + return { ...block, text: redacted }; + } + } + return block; + }); + return (changed ? { ...toolResultMsg, content } : toolResultMsg) as Message; + } + + if (msg.role === "assistant") { + const assistantMsg = msg as AssistantMessage; + let changed = false; + const content = assistantMsg.content.map((block): AssistantMessage["content"][number] => { + if (block.type === "text") { + const redacted = redactSensitiveCredentials(block.text); + if (redacted !== block.text) { + changed = true; + return { ...block, text: redacted }; + } + } else if (block.type === "thinking") { + const redacted = redactSensitiveCredentials(block.thinking); + if (redacted !== block.thinking) { + changed = true; + return { ...block, thinking: redacted, thinkingSignature: undefined }; + } + } else if (block.type === "toolCall") { + if (block.arguments) { + const { result: redactedArgs, changed: argsChanged } = redactSensitiveInObject(block.arguments); + if (argsChanged) { + changed = true; + const castArgs = + redactedArgs && typeof redactedArgs === "object" && !Array.isArray(redactedArgs) + ? (redactedArgs as Record) + : undefined; + return { + ...block, + arguments: castArgs, + thoughtSignature: undefined, + } as AssistantMessage["content"][number]; + } + } + } + return block; + }); + return (changed ? { ...assistantMsg, content } : assistantMsg) as Message; + } + + return msg; + }); +} + export function transformMessages( messages: Message[], model: Model, @@ -294,6 +453,10 @@ export function transformMessages( duplicateToolCallIdSuffixPrefix = "_dup", targetCompat: Model["compat"] = model.compat, ): Message[] { + // Redact sensitive credential-like patterns from all outbound messages + // to prevent security block errors from LLM providers (e.g. invalid_prompt). + messages = redactSensitiveCredentialsInMessages(messages); + // Drop assistant `toolCall` blocks with empty/whitespace `id` or `name` // (and their matched `toolResult` messages) before anything else looks at // the history. Replays of these would 400 every provider — see @@ -554,6 +717,13 @@ export function transformMessages( return []; } + if (block.type === "image") { + // Assistant images are display artifacts. No provider accepts them + // in an assistant replay turn; the native Responses result remains + // in providerPayload for OpenAI replay. + return []; + } + if (block.type === "text") { if (isSameModel) return block; return { diff --git a/packages/ai/src/registry/alibaba-coding-plan.ts b/packages/ai/src/registry/alibaba-coding-plan.ts index dee087c39..6fcda83ce 100644 --- a/packages/ai/src/registry/alibaba-coding-plan.ts +++ b/packages/ai/src/registry/alibaba-coding-plan.ts @@ -71,13 +71,22 @@ export async function loginAlibabaCodingPlan(options: OAuthController): Promise< } options.onProgress?.("Validating API key..."); - await apiKeyValidation.validateOpenAICompatibleApiKey({ - provider: "Alibaba Coding Plan", - apiKey: trimmed, - baseUrl, - model: VALIDATION_MODEL, - signal: options.signal, - }); + if (choice === "3") { + await apiKeyValidation.validateApiKeyAgainstModelsEndpoint({ + provider: "Alibaba Coding Plan", + apiKey: trimmed, + modelsUrl: `${baseUrl}/models`, + signal: options.signal, + }); + } else { + await apiKeyValidation.validateOpenAICompatibleApiKey({ + provider: "Alibaba Coding Plan", + apiKey: trimmed, + baseUrl, + model: VALIDATION_MODEL, + signal: options.signal, + }); + } return { access: trimmed, diff --git a/packages/ai/src/registry/api-key-login.ts b/packages/ai/src/registry/api-key-login.ts index 94783dbb4..f202eab63 100644 --- a/packages/ai/src/registry/api-key-login.ts +++ b/packages/ai/src/registry/api-key-login.ts @@ -31,7 +31,7 @@ type AnthropicMessagesValidation = { type ModelsEndpointValidation = { kind: "models-endpoint"; provider: string; - modelsUrl: string; + modelsUrl: string | (() => string); headers?: Record | (() => Record | undefined); }; @@ -99,7 +99,10 @@ export function createApiKeyLogin(config: ApiKeyLoginConfig): (options: OAuthCon await validateApiKeyAgainstModelsEndpoint({ provider: config.validation.provider, apiKey: trimmed, - modelsUrl: config.validation.modelsUrl, + modelsUrl: + typeof config.validation.modelsUrl === "function" + ? config.validation.modelsUrl() + : config.validation.modelsUrl, headers: config.validation.headers, signal: options.signal, fetch: options.fetch, diff --git a/packages/ai/src/registry/moonshot.ts b/packages/ai/src/registry/moonshot.ts index 7b38a541c..749ef4047 100644 --- a/packages/ai/src/registry/moonshot.ts +++ b/packages/ai/src/registry/moonshot.ts @@ -1,7 +1,13 @@ +import { $env } from "@oh-my-pi/pi-utils"; import { createApiKeyLogin } from "./api-key-login"; import type { OAuthLoginCallbacks } from "./oauth/types"; import type { ProviderDefinition } from "./types"; +function resolveMoonshotModelsUrl(): string { + const baseUrl = $env.MOONSHOT_BASE_URL?.trim() || "https://api.moonshot.ai/v1"; + return `${baseUrl.replace(/\/+$/, "")}/models`; +} + export const loginMoonshot = createApiKeyLogin({ providerLabel: "Moonshot", authUrl: "https://platform.moonshot.ai/console/api-keys", @@ -11,7 +17,7 @@ export const loginMoonshot = createApiKeyLogin({ validation: { kind: "models-endpoint", provider: "moonshot", - modelsUrl: "https://api.moonshot.ai/v1/models", + modelsUrl: resolveMoonshotModelsUrl, }, }); diff --git a/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts b/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts index ac5012ba0..66eb3109d 100644 --- a/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts +++ b/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts @@ -1,19 +1,35 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; -import { isXAIAccessTokenExpiring, loginXAIOAuth, refreshXAIOAuthToken, validateXAIEndpoint } from "../xai-oauth"; +import { + buildXAICliBillingUrl, + extractXAIAccessTokenSubject, + fetchXAIOAuthIdentity, + getXAICliBillingHeaders, + isXAIAccessTokenExpiring, + loginXAIOAuth, + parseXAIAccessTokenPayload, + refreshXAIOAuthToken, + validateXAIBillingEndpoint, + validateXAIEndpoint, +} from "../xai-oauth"; afterEach(() => { vi.restoreAllMocks(); }); function jwtWithExp(exp: number): string { + return jwtWithPayload({ exp }); +} + +function jwtWithPayload(payload: Record): string { const header = Buffer.from(JSON.stringify({ alg: "HS256", typ: "JWT" })).toString("base64url"); - const payload = Buffer.from(JSON.stringify({ exp })).toString("base64url"); - return `${header}.${payload}.sig`; + const encodedPayload = Buffer.from(JSON.stringify(payload)).toString("base64url"); + return `${header}.${encodedPayload}.sig`; } const DISCOVERY_URL = "https://auth.x.ai/.well-known/openid-configuration"; const DEVICE_CODE_URL = "https://auth.x.ai/oauth2/device/code"; const TOKEN_ENDPOINT = "https://auth.x.ai/oauth2/token"; +const USERINFO_URL = "https://auth.x.ai/oauth2/userinfo"; const CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"; const SCOPE = "openid profile email offline_access grok-cli:access api:access"; @@ -43,7 +59,10 @@ function jsonResponse(body: unknown, status: number = 200): Response { }); } -function createDeviceFlowFetch(tokenResponses: readonly TokenResponse[]) { +function createDeviceFlowFetch( + tokenResponses: readonly TokenResponse[], + userinfoResponse: TokenResponse = { body: {} }, +) { const requests: RecordedRequest[] = []; let tokenResponseIndex = 0; const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { @@ -64,6 +83,9 @@ function createDeviceFlowFetch(tokenResponses: readonly TokenResponse[]) { } return jsonResponse(tokenResponse.body, tokenResponse.status); } + if (url === USERINFO_URL) { + return jsonResponse(userinfoResponse.body, userinfoResponse.status); + } throw new Error(`Unexpected xAI OAuth request: ${url}`); }); @@ -101,6 +123,109 @@ describe("isXAIAccessTokenExpiring", () => { }); }); +describe("xAI OAuth helpers", () => { + it("parses JWT payloads and extracts subjects", () => { + const token = jwtWithPayload({ sub: " subject-123 ", exp: 1_900_000_000 }); + + expect(parseXAIAccessTokenPayload(token)).toEqual({ sub: " subject-123 ", exp: 1_900_000_000 }); + expect(extractXAIAccessTokenSubject(token)).toBe("subject-123"); + expect(parseXAIAccessTokenPayload("not-a-jwt")).toBeNull(); + expect(extractXAIAccessTokenSubject("not-a-jwt")).toBeUndefined(); + }); + + it("builds the billing URL and CLI-aligned headers", () => { + expect(buildXAICliBillingUrl()).toBe("https://cli-chat-proxy.grok.com/v1/billing?format=credits"); + expect(buildXAICliBillingUrl("tokens")).toBe("https://cli-chat-proxy.grok.com/v1/billing?format=tokens"); + expect(getXAICliBillingHeaders({ accessToken: "access-token" })).toEqual({ + Authorization: "Bearer access-token", + Accept: "application/json", + "X-XAI-Token-Auth": "xai-grok-cli", + }); + }); + + it("pins SuperGrok billing URLs to https grok.com hosts", () => { + expect(validateXAIBillingEndpoint("https://cli-chat-proxy.grok.com/v1/billing")).toBe( + "https://cli-chat-proxy.grok.com/v1/billing", + ); + expect(() => validateXAIBillingEndpoint("https://auth.x.ai/v1/billing")).toThrow(/Invalid xAI billing_url/); + expect(() => validateXAIBillingEndpoint("http://cli-chat-proxy.grok.com/v1/billing")).toThrow( + /Invalid xAI billing_url/, + ); + expect(() => validateXAIBillingEndpoint("https://evil.com/v1/billing")).toThrow(/Invalid xAI billing_url/); + }); + + it("normalizes OIDC userinfo identity", async () => { + const requests: RecordedRequest[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + requests.push({ + url: typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(), + init, + }); + return jsonResponse({ sub: "profile-sub", email: "User@Example.com", name: "User" }); + }); + + await expect(fetchXAIOAuthIdentity("access-token", fetchMock as unknown as typeof fetch)).resolves.toEqual({ + accountId: "profile-sub", + email: "user@example.com", + name: "User", + }); + expect(requests[0]?.url).toBe(USERINFO_URL); + expect(new Headers(requests[0]?.init?.headers)).toEqual( + new Headers({ Authorization: "Bearer access-token", Accept: "application/json" }), + ); + expect(requests[0]?.init?.redirect).toBe("error"); + }); + + it("combines caller cancellation with the 15-second userinfo timeout", async () => { + const timeoutControllers: AbortController[] = []; + const timeoutSpy = vi.spyOn(AbortSignal, "timeout").mockImplementation(timeoutMs => { + expect(timeoutMs).toBe(15_000); + const controller = new AbortController(); + timeoutControllers.push(controller); + return controller.signal; + }); + const requests: RecordedRequest[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + requests.push({ + url: typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(), + init, + }); + const { promise, reject } = Promise.withResolvers(); + const requestSignal = init?.signal; + if (!requestSignal) { + reject(new Error("expected userinfo request signal")); + } else if (requestSignal.aborted) { + reject(requestSignal.reason); + } else { + requestSignal.addEventListener("abort", () => reject(requestSignal.reason), { once: true }); + } + return promise; + }); + + const callerController = new AbortController(); + const callerCancelled = fetchXAIOAuthIdentity( + "access-token", + fetchMock as unknown as typeof fetch, + callerController.signal, + ); + callerController.abort(); + await expect(callerCancelled).resolves.toBeNull(); + expect(requests[0]?.init?.signal).not.toBe(callerController.signal); + + expect(timeoutSpy).toHaveBeenCalledWith(15_000); + const timeoutCancelled = fetchXAIOAuthIdentity( + "access-token", + fetchMock as unknown as typeof fetch, + new AbortController().signal, + ); + const timeoutController = timeoutControllers[1]; + expect(timeoutController).toBeDefined(); + timeoutController?.abort(); + await expect(timeoutCancelled).resolves.toBeNull(); + expect(requests[1]?.init?.signal).not.toBe(timeoutController?.signal); + }); +}); + describe("validateXAIEndpoint", () => { it("rejects non-HTTPS URLs", () => { expect(() => validateXAIEndpoint("http://x.ai/token", "token_endpoint")).toThrow(/Invalid xAI token_endpoint/); @@ -131,6 +256,35 @@ describe("refreshXAIOAuthToken", () => { ); expect(fetchMock).not.toHaveBeenCalled(); }); + + it("persists refreshed OAuth identity from OIDC userinfo", async () => { + const accessToken = jwtWithPayload({ sub: "jwt-sub" }); + const requests: RecordedRequest[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(); + requests.push({ url, init }); + if (url === DISCOVERY_URL) return jsonResponse({ token_endpoint: TOKEN_ENDPOINT }); + if (url === TOKEN_ENDPOINT) { + return jsonResponse({ + access_token: accessToken, + expires_in: 3600, + }); + } + if (url === USERINFO_URL) return jsonResponse({ sub: "profile-sub", email: "User@Example.com" }); + throw new Error(`Unexpected xAI OAuth request: ${url}`); + }); + + await expect( + refreshXAIOAuthToken("old-refresh-token", fetchMock as unknown as typeof fetch), + ).resolves.toMatchObject({ + access: accessToken, + refresh: "old-refresh-token", + accountId: "profile-sub", + email: "user@example.com", + }); + expect(requests.map(request => request.url)).toEqual([DISCOVERY_URL, TOKEN_ENDPOINT, USERINFO_URL]); + expect(requests.find(request => request.url === TOKEN_ENDPOINT)?.init?.redirect).toBe("error"); + }); }); describe("loginXAIOAuth", () => { @@ -165,7 +319,12 @@ describe("loginXAIOAuth", () => { onManualCodeInput, }); - expect(requests.map(request => request.url)).toEqual([DISCOVERY_URL, DEVICE_CODE_URL, TOKEN_ENDPOINT]); + expect(requests.map(request => request.url)).toEqual([ + DISCOVERY_URL, + DEVICE_CODE_URL, + TOKEN_ENDPOINT, + USERINFO_URL, + ]); const discoveryRequest = requests[0]; expect(discoveryRequest?.init?.method).toBe("GET"); @@ -230,6 +389,7 @@ describe("loginXAIOAuth", () => { const tokenRequests = requests.filter(request => request.url === TOKEN_ENDPOINT); expect(tokenRequests).toHaveLength(3); + expect(tokenRequests.every(request => request.init?.redirect === "error")).toBe(true); expect(tokenRequests.map(request => Object.fromEntries(requestForm(request)))).toEqual([ { grant_type: "urn:ietf:params:oauth:grant-type:device_code", @@ -252,6 +412,54 @@ describe("loginXAIOAuth", () => { expect(credentials.refresh).toBe("eventual-refresh-token"); }); + it("retains the JWT subject and succeeds when OIDC userinfo fails", async () => { + const accessToken = jwtWithPayload({ sub: "jwt-sub" }); + const controller = new AbortController(); + const { fetchMock, requests } = createDeviceFlowFetch( + [ + { + body: { + access_token: accessToken, + refresh_token: "refresh-token", + expires_in: 3600, + }, + }, + ], + { body: { error: "userinfo unavailable" }, status: 503 }, + ); + + const credentials = await loginXAIOAuth({ fetch: fetchMock, signal: controller.signal }); + + expect(credentials).toMatchObject({ + access: accessToken, + refresh: "refresh-token", + accountId: "jwt-sub", + }); + expect(requests.at(-1)?.url).toBe(USERINFO_URL); + expect(requests.at(-1)?.init?.signal).not.toBe(controller.signal); + }); + + it("keeps the JWT subject when userinfo returns only an email", async () => { + const accessToken = jwtWithPayload({ sub: "jwt-sub" }); + const { fetchMock } = createDeviceFlowFetch( + [ + { + body: { + access_token: accessToken, + refresh_token: "refresh-token", + expires_in: 3600, + }, + }, + ], + { body: { email: "User@Example.com" } }, + ); + + await expect(loginXAIOAuth({ fetch: fetchMock })).resolves.toMatchObject({ + accountId: "jwt-sub", + email: "user@example.com", + }); + }); + it("rejects a token response that omits access_token", async () => { const { fetchMock, requests } = createDeviceFlowFetch([ { diff --git a/packages/ai/src/registry/oauth/kimi.ts b/packages/ai/src/registry/oauth/kimi.ts index 35423f4dc..2041a3eb0 100644 --- a/packages/ai/src/registry/oauth/kimi.ts +++ b/packages/ai/src/registry/oauth/kimi.ts @@ -7,7 +7,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { scheduler } from "node:timers/promises"; -import { $env, getAgentDir, isEnoent } from "@oh-my-pi/pi-utils"; +import { $env, getAgentDir } from "@oh-my-pi/pi-utils"; import packageJson from "../../../package.json" with { type: "json" }; import * as AIError from "../../error"; import type { OAuthController, OAuthCredentials } from "./types"; @@ -57,21 +57,29 @@ function getDeviceModel(): string { return formatDeviceModel(label, release, arch); } +// Device id identifies this install to Kimi. Persistence is best-effort: a +// missing/unwritable agent dir must never break header construction (and with +// it every usage probe / request that spreads getKimiCommonHeaders()) — fall +// back to a per-process ephemeral id instead. let getDeviceId = (): string => { const deviceIdPath = path.join(getAgentDir(), DEVICE_ID_FILENAME); try { - const existing = fs.readFileSync(deviceIdPath, "utf-8"); - const trimmed = existing.trim(); - if (trimmed) { - getDeviceId = () => trimmed; - return trimmed; + const existing = fs.readFileSync(deviceIdPath, "utf-8").trim(); + if (existing) { + getDeviceId = () => existing; + return existing; } - } catch (error) { - if (!isEnoent(error)) throw error; + } catch { + // Unreadable device-id file: regenerate below. } const deviceId = crypto.randomUUID().replace(/-/g, ""); - fs.writeFileSync(deviceIdPath, `${deviceId}\n`, { mode: 0o600 }); + try { + fs.mkdirSync(path.dirname(deviceIdPath), { recursive: true }); + fs.writeFileSync(deviceIdPath, `${deviceId}\n`, { mode: 0o600 }); + } catch { + // Persist failure → ephemeral id for this process. + } getDeviceId = () => deviceId; return deviceId; }; diff --git a/packages/ai/src/registry/oauth/openai-codex.ts b/packages/ai/src/registry/oauth/openai-codex.ts index 725161aab..63e7be5c5 100644 --- a/packages/ai/src/registry/oauth/openai-codex.ts +++ b/packages/ai/src/registry/oauth/openai-codex.ts @@ -31,6 +31,7 @@ const DEVICE_MAX_POLLS = 120; type JwtPayload = { [JWT_CLAIM_PATH]?: { chatgpt_account_id?: string; + chatgpt_plan_type?: string; }; [JWT_PROFILE_CLAIM]?: { email?: string; @@ -50,14 +51,27 @@ export function decodeJwt>(token: string): T | null } } -function getTokenProfile(accessToken: string): { accountId?: string; email?: string } { +/** + * Identity slice decoded from the token claims. The ChatGPT workspace + * (`chatgpt_account_id`) is the subscription pool the token draws limits + * from — one account email can hold several (e.g. a personal Pro plan plus a + * Team seat). `chatgpt_plan_type` may only be present on the `id_token`. + */ +function getTokenProfile( + accessToken: string, + idToken?: string, +): { accountId?: string; email?: string; planType?: string } { const payload = decodeJwt(accessToken); + const idPayload = idToken ? decodeJwt(idToken) : null; const auth = payload?.[JWT_CLAIM_PATH]; + const idAuth = idPayload?.[JWT_CLAIM_PATH]; const accountId = auth?.chatgpt_account_id; const email = payload?.[JWT_PROFILE_CLAIM]?.email?.trim().toLowerCase(); + const planType = (auth?.chatgpt_plan_type ?? idAuth?.chatgpt_plan_type)?.trim().toLowerCase(); return { accountId: typeof accountId === "string" && accountId.length > 0 ? accountId : undefined, email: typeof email === "string" && email.length > 0 ? email : undefined, + planType: typeof planType === "string" && planType.length > 0 ? planType : undefined, }; } @@ -185,6 +199,7 @@ async function exchangeCodeForToken( const tokenData = (await tokenResponse.json()) as { access_token?: string; refresh_token?: string; + id_token?: string; expires_in?: number; }; @@ -192,7 +207,7 @@ async function exchangeCodeForToken( throw new AIError.OAuthError("Token response missing required fields", { kind: "validation" }); } - const { accountId, email } = getTokenProfile(tokenData.access_token); + const { accountId, email, planType } = getTokenProfile(tokenData.access_token, tokenData.id_token); if (!accountId) { throw new AIError.OAuthError("Failed to extract accountId from token", { kind: "validation" }); } @@ -203,6 +218,8 @@ async function exchangeCodeForToken( expires: Date.now() + tokenData.expires_in * 1000, accountId, email, + orgId: accountId, + orgName: planType, }; } @@ -354,6 +371,9 @@ export async function refreshOpenAICodexToken(refreshToken: string): Promise { * @throws Error with message `Invalid xAI : ` when the URL fails * either scheme or host validation. */ +function isXaiAuthHostname(host: string): boolean { + return host === "x.ai" || host.endsWith(".x.ai"); +} + +/** SuperGrok CLI billing proxy host (`cli-chat-proxy.grok.com`), not the OIDC issuer. */ +function isXaiBillingHostname(host: string): boolean { + return host === "grok.com" || host.endsWith(".grok.com"); +} + export function validateXAIEndpoint(url: string, field: string): string { let parsed: URL; try { @@ -62,7 +75,28 @@ export function validateXAIEndpoint(url: string, field: string): string { throw new AIError.OAuthError(`Invalid xAI ${field}: ${url}`, { kind: "validation", provider: "xai" }); } const host = parsed.hostname.toLowerCase(); - if (!host || (host !== "x.ai" && !host.endsWith(".x.ai"))) { + if (!host || !isXaiAuthHostname(host)) { + throw new AIError.OAuthError(`Invalid xAI ${field}: ${url}`, { kind: "validation", provider: "xai" }); + } + return url; +} + +/** + * Pin SuperGrok billing URLs to HTTPS `grok.com` / `*.grok.com`. + * The CLI billing proxy is intentionally not on `*.x.ai`. + */ +export function validateXAIBillingEndpoint(url: string, field: string = "billing_url"): string { + let parsed: URL; + try { + parsed = new URL(url); + } catch { + throw new AIError.OAuthError(`Invalid xAI ${field}: ${url}`, { kind: "validation", provider: "xai" }); + } + if (parsed.protocol !== "https:") { + throw new AIError.OAuthError(`Invalid xAI ${field}: ${url}`, { kind: "validation", provider: "xai" }); + } + const host = parsed.hostname.toLowerCase(); + if (!host || !isXaiBillingHostname(host)) { throw new AIError.OAuthError(`Invalid xAI ${field}: ${url}`, { kind: "validation", provider: "xai" }); } return url; @@ -124,6 +158,22 @@ async function xaiOAuthDiscovery( return { token_endpoint: tokenEndpoint }; } +/** Decode an xAI access-token JWT payload without verifying its signature. */ +export function parseXAIAccessTokenPayload(jwt: string): Record | null { + try { + if (typeof jwt !== "string" || !jwt.includes(".")) return null; + const parts = jwt.split("."); + if (parts.length < 2) return null; + const payloadPart = parts[1]; + if (!payloadPart) return null; + const decoded = Buffer.from(payloadPart, "base64url").toString("utf8"); + const payload = JSON.parse(decoded) as unknown; + return isRecord(payload) && !Array.isArray(payload) ? payload : null; + } catch { + return null; + } +} + /** * Check whether a JWT access token is at or past its `exp` claim (with an * optional refresh-skew margin). @@ -132,25 +182,100 @@ async function xaiOAuthDiscovery( * not token validation. */ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): boolean { + const payload = parseXAIAccessTokenPayload(jwt); + if (!payload) return false; + const exp = payload.exp; + if (typeof exp !== "number" || !Number.isFinite(exp)) return false; + const now = Math.floor(Date.now() / 1000); + const skew = Math.max(0, Math.floor(skewSeconds)); + return exp <= now + skew; +} + +/** Extract the stable xAI subject UUID from an access token. */ +export function extractXAIAccessTokenSubject(jwt: string): string | undefined { + const sub = parseXAIAccessTokenPayload(jwt)?.sub; + return typeof sub === "string" && sub.trim() ? sub.trim() : undefined; +} + +export interface XAIOAuthIdentity { + accountId?: string; + email?: string; + name?: string; +} + +/** Fetch optional OIDC userinfo for a valid xAI access token. */ +export async function fetchXAIOAuthIdentity( + accessToken: string, + fetchOverride?: FetchImpl, + signal?: AbortSignal, +): Promise { + const token = accessToken.trim(); + if (!token) return null; + const fetchImpl = fetchOverride ?? fetch; try { - if (typeof jwt !== "string" || !jwt.includes(".")) return false; - const parts = jwt.split("."); - if (parts.length < 2) return false; - const payloadPart = parts[1]; - if (!payloadPart) return false; - const decoded = Buffer.from(payloadPart, "base64url").toString("utf8"); - const payload: unknown = JSON.parse(decoded); - if (!isRecord(payload)) return false; - const exp = payload.exp; - if (typeof exp !== "number" || !Number.isFinite(exp)) return false; - const now = Math.floor(Date.now() / 1000); - const skew = Math.max(0, Math.floor(skewSeconds)); - return exp <= now + skew; + const response = await fetchImpl(XAI_OAUTH_USERINFO_URL, { + method: "GET", + headers: { + Authorization: `Bearer ${token}`, + Accept: "application/json", + }, + redirect: "error", + signal: signal + ? AbortSignal.any([signal, AbortSignal.timeout(DISCOVERY_TIMEOUT_MS)]) + : AbortSignal.timeout(DISCOVERY_TIMEOUT_MS), + }); + if (!response.ok) return null; + const payload = (await response.json()) as unknown; + if (!isRecord(payload) || Array.isArray(payload)) return null; + const sub = typeof payload.sub === "string" && payload.sub.trim() ? payload.sub.trim() : undefined; + const email = typeof payload.email === "string" && payload.email.trim() ? payload.email.trim() : undefined; + const name = typeof payload.name === "string" && payload.name.trim() ? payload.name.trim() : undefined; + if (!sub && !email && !name) return null; + return { + ...(sub ? { accountId: sub } : {}), + ...(email ? { email: email.toLowerCase() } : {}), + ...(name ? { name } : {}), + }; } catch { - return false; + return null; } } +async function withXAIOAuthIdentity( + credentials: OAuthCredentials, + fetchOverride?: FetchImpl, + signal?: AbortSignal, +): Promise { + const identity = await fetchXAIOAuthIdentity(credentials.access, fetchOverride, signal); + const accountId = identity?.accountId ?? credentials.accountId ?? extractXAIAccessTokenSubject(credentials.access); + const email = identity?.email ?? credentials.email; + return { + ...credentials, + ...(accountId ? { accountId } : {}), + ...(email ? { email } : {}), + }; +} + +/** Build the SuperGrok CLI billing URL. */ +export function buildXAICliBillingUrl(format: string = XAI_CLI_BILLING_FORMAT): string { + const url = new URL(XAI_CLI_BILLING_PATH, XAI_CLI_BILLING_BASE_URL); + url.searchParams.set("format", format); + return validateXAIBillingEndpoint(url.toString()); +} + +/** + * Headers for SuperGrok CLI billing (`cli-chat-proxy.grok.com`). + * Official Grok CLI also sends `X-XAI-Token-Auth: xai-grok-cli` on this host; + * include it so billing stays on the same product gate as chat inference. + */ +export function getXAICliBillingHeaders(options: { accessToken: string }): Record { + return { + Authorization: `Bearer ${options.accessToken}`, + Accept: "application/json", + "X-XAI-Token-Auth": "xai-grok-cli", + }; +} + function parseXAIDeviceAuthorization(payload: unknown): XAIDeviceAuthorization { if (!isRecord(payload)) { throw new AIError.OAuthError("xAI device-code response was not a JSON object.", { @@ -304,6 +429,7 @@ async function pollXAIDeviceToken( client_id: XAI_OAUTH_CLIENT_ID, device_code: deviceCode, }), + redirect: "error", signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal, }); } catch (error) { @@ -364,12 +490,13 @@ export async function loginXAIOAuth(ctrl: OAuthController): Promise pollXAIDeviceToken(discovery.token_endpoint, device.deviceCode, fetchImpl, ctrl.signal), intervalSeconds: device.intervalSeconds, expiresInSeconds: device.expiresInSeconds, signal: ctrl.signal, }); + return withXAIOAuthIdentity(credentials, fetchImpl, ctrl.signal); } /** @@ -400,6 +527,7 @@ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: Accept: "application/json", }, body, + redirect: "error", signal: AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS), }); @@ -426,5 +554,6 @@ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: { kind: "validation", provider: "xai", cause: error }, ); } - return parseXAITokenResponse(payload, "xAI token refresh response", refreshToken); + const credentials = parseXAITokenResponse(payload, "xAI token refresh response", refreshToken); + return withXAIOAuthIdentity(credentials, fetchImpl); } diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 32329ec4c..94c74c94e 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -20,6 +20,7 @@ import { getCustomApi } from "./api-registry"; import { createAuthRetryKeyState, isApiKeyResolver, resolveNextAuthRetryKey } from "./auth-retry"; import * as AIError from "./error"; import { ProviderHttpError } from "./error"; +import { isInvalidatedOAuthTokenError } from "./error/auth-classify"; import { isUsageLimitOutcome } from "./error/rate-limit"; import type { BedrockOptions } from "./providers/amazon-bedrock"; import type { AnthropicOptions } from "./providers/anthropic"; @@ -979,6 +980,7 @@ function isRetryableUpstreamError(error: unknown, status: number | undefined, me // `parseRateLimitReason` and stay in the provider's own backoff layer // instead of burning siblings. if (AIError.isUsageLimit(error)) return true; + if (isInvalidatedOAuthTokenError(error)) return true; if (status === 401) return true; return isUsageLimitOutcome(status, message); } @@ -1176,12 +1178,17 @@ export function streamSimple( // Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API if (isKimiModel(model)) { - // Pass raw SimpleStreamOptions - streamKimi handles mapping internally - return withProviderInFlightLimit(model, requestOptions, () => + // streamKimi handles openai/anthropic format mapping internally, but the + // mandatory-reasoning clamp is a request-shaping concern owned here: K3's + // `supports_thinking_type: "only"` endpoint rejects disabled/omitted + // thinking, so clamp disabled requests to the lowest supported effort + // (mirrors the mapOptionsForApi path every other provider takes). + const kimiOptions = normalizeMandatoryReasoningOptions(model, requestOptions); + return withProviderInFlightLimit(model, kimiOptions, () => streamKimi(model as Model<"openai-completions">, context, { - ...requestOptions, + ...kimiOptions, apiKey, - format: requestOptions?.kimiApiFormat ?? "anthropic", + format: kimiOptions?.kimiApiFormat, }), ); } diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index d5903e5dd..773b43361 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -141,13 +141,17 @@ function isOpenAIServiceTierApi(api: Api | undefined): boolean { return api === "openai-completions" || api === "openai-responses" || api === "openai-codex-responses"; } -function hasDedicatedServiceTierControl(provider: Provider | undefined): boolean { - return provider === "fireworks"; +function excludesInferredOpenAIServiceTier(provider: Provider | undefined): boolean { + // Fireworks has its own priority-only control. GitHub Copilot proxies OpenAI + // models but rejects OpenAI's `service_tier` request field. + return provider === "fireworks" || provider === "github-copilot"; } function isOpenAIServiceTierModel(model: ServiceTierModel): boolean { return ( - !hasDedicatedServiceTierControl(model.provider) && isOpenAIServiceTierApi(model.api) && isOpenAIModelId(model.id) + !excludesInferredOpenAIServiceTier(model.provider) && + isOpenAIServiceTierApi(model.api) && + isOpenAIModelId(model.id) ); } @@ -159,7 +163,8 @@ function isOpenAIServiceTierModel(model: ServiceTierModel): boolean { * `openai/`); Claude on Bedrock/Vertex (api `anthropic-messages`) is the * anthropic family even though its provider is `amazon-bedrock`/`google-vertex`. * Custom OpenAI-compatible relays that serve OpenAI model ids are OpenAI family - * too unless that provider owns a separate tier control such as Fireworks. + * too unless the provider owns a separate tier control (Fireworks) or rejects + * OpenAI's service-tier field (GitHub Copilot). */ export function serviceTierFamily(model: ServiceTierModel): ServiceTierFamily | undefined { const provider = model.provider; @@ -539,7 +544,7 @@ export interface SimpleStreamOptions extends Omit { toolChoice?: ToolChoice; /** OpenAI service tier for processing priority/cost control. Ignored by non-OpenAI providers. */ serviceTier?: ServiceTier; - /** API format for Kimi Code provider: "openai" or "anthropic" (default: "anthropic") */ + /** Explicit Kimi Code API format override; omitted uses live per-model protocol metadata. */ kimiApiFormat?: "openai" | "anthropic"; /** API format for Synthetic provider: "openai" or "anthropic" (default: "openai") */ syntheticApiFormat?: "openai" | "anthropic"; @@ -706,7 +711,14 @@ export interface ContextSnapshot { export interface AssistantMessage { role: "assistant"; - content: (TextContent | ThinkingContent | RedactedThinkingContent | AnthropicFallbackContent | ToolCall)[]; + content: ( + | TextContent + | ThinkingContent + | RedactedThinkingContent + | AnthropicFallbackContent + | ImageContent + | ToolCall + )[]; api: Api; provider: Provider; model: string; @@ -893,6 +905,7 @@ export type AssistantMessageEvent = | { type: "thinking_start"; contentIndex: number; partial: AssistantMessage } | { type: "thinking_delta"; contentIndex: number; delta: string; partial: AssistantMessage } | { type: "thinking_end"; contentIndex: number; content: string; partial: AssistantMessage } + | { type: "image_end"; contentIndex: number; content: ImageContent; partial: AssistantMessage } | { type: "toolcall_start"; contentIndex: number; partial: AssistantMessage } | { type: "toolcall_delta"; contentIndex: number; delta: string; partial: AssistantMessage } | { type: "toolcall_end"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage } diff --git a/packages/ai/src/usage.ts b/packages/ai/src/usage.ts index a04979fbf..4c7cb85cf 100644 --- a/packages/ai/src/usage.ts +++ b/packages/ai/src/usage.ts @@ -20,6 +20,12 @@ export interface UsageWindow { durationMs?: number; /** Absolute reset timestamp in milliseconds since epoch. */ resetsAt?: number; + /** + * Verb rendered before the {@link resetsAt} countdown (e.g. "tick", "regen"). + * Defaults to "resets" — override for rolling windows where the timestamp is + * an incremental regeneration step rather than a full window reset. + */ + resetLabel?: string; } /** Quantitative usage data. */ @@ -189,6 +195,7 @@ export const usageWindowSchema = type({ label: "string", "durationMs?": "number", "resetsAt?": "number", + "resetLabel?": "string", }); export const usageAmountSchema = type({ diff --git a/packages/ai/src/usage/claude.ts b/packages/ai/src/usage/claude.ts index c346115ee..fbcf5cd82 100644 --- a/packages/ai/src/usage/claude.ts +++ b/packages/ai/src/usage/claude.ts @@ -96,11 +96,6 @@ interface ParsedApiLimitEntry { displayName?: string; } -type ClaudeUsagePayload = { - payload: ClaudeUsageResponse; - orgId?: string; -}; - function parseIsoTime(value: string | undefined): number | undefined { if (!value) return undefined; const parsed = Date.parse(value); @@ -178,22 +173,17 @@ function getNestedPayloadString(payload: Record, key: string, n return isRecord(nested) ? getPayloadString(nested, nestedKey) : undefined; } -function extractUsageIdentity(payload: ClaudeUsageResponse, orgId?: string): { accountId?: string; email?: string } { - if (!isRecord(payload)) return { accountId: orgId }; +function extractUsageIdentity(payload: ClaudeUsageResponse): { accountId?: string; email?: string } { + if (!isRecord(payload)) return {}; const accountId = getPayloadString(payload, "account_id") ?? getPayloadString(payload, "accountId") ?? getPayloadString(payload, "user_id") ?? getPayloadString(payload, "userId") ?? - getPayloadString(payload, "org_id") ?? - getPayloadString(payload, "orgId") ?? getNestedPayloadString(payload, "account", "uuid") ?? getNestedPayloadString(payload, "account", "id") ?? - getNestedPayloadString(payload, "organization", "uuid") ?? - getNestedPayloadString(payload, "organization", "id") ?? getNestedPayloadString(payload, "user", "uuid") ?? - getNestedPayloadString(payload, "user", "id") ?? - orgId; + getNestedPayloadString(payload, "user", "id"); const email = getPayloadString(payload, "email") ?? getPayloadString(payload, "user_email") ?? @@ -263,16 +253,13 @@ async function fetchUsagePayload( headers: Record, ctx: UsageFetchContext, signal?: AbortSignal, -): Promise { +): Promise { if (signal?.aborted) return null; let lastPayload: ClaudeUsageResponse | null = null; - let lastOrgId: string | undefined; for (let attempt = 0; attempt < MAX_ATTEMPTS; attempt++) { try { const response = await ctx.fetch(url, { headers, signal }); - const orgId = response.headers.get("anthropic-organization-id")?.trim() || undefined; - lastOrgId = orgId ?? lastOrgId; if (!response.ok) { const retryable = isRetryableStatus(response.status); @@ -292,7 +279,7 @@ async function fetchUsagePayload( if (isRecord(parsed)) { const payload = parsed as ClaudeUsageResponse; lastPayload = payload; - if (hasUsageData(payload)) return { payload, orgId }; + if (hasUsageData(payload)) return payload; } ctx.logger?.warn("Claude usage response missing usage data", { @@ -311,7 +298,7 @@ async function fetchUsagePayload( } } - return lastPayload ? { payload: lastPayload, orgId: lastOrgId } : null; + return lastPayload; } interface ClaudeProfile { @@ -507,9 +494,8 @@ async function fetchClaudeUsage(params: UsageFetchParams, ctx: UsageFetchContext authorization: `Bearer ${credential.accessToken}`, }; - const payloadResult = await fetchUsagePayload(url, headers, ctx, params.signal); - if (!payloadResult || !isRecord(payloadResult.payload)) return null; - const { payload, orgId } = payloadResult; + const payload = await fetchUsagePayload(url, headers, ctx, params.signal); + if (!payload || !isRecord(payload)) return null; const apiLimitEntries = parseApiLimitEntries(payload.limits); const fiveHour = parseBucket(payload.five_hour) ?? apiLimitEntries.find(entry => entry.kind === "session")?.bucket; @@ -563,7 +549,7 @@ async function fetchClaudeUsage(params: UsageFetchParams, ctx: UsageFetchContext ].filter((limit): limit is UsageLimit => limit !== null); if (limits.length === 0) return null; - const identity = extractUsageIdentity(payload, orgId); + const identity = extractUsageIdentity(payload); let accountId = identity.accountId ?? credential.accountId; let email = identity.email ?? credential.email; if ((!accountId || !email) && !params.signal?.aborted) { @@ -580,7 +566,7 @@ async function fetchClaudeUsage(params: UsageFetchParams, ctx: UsageFetchContext endpoint: url, ...(accountId ? { accountId } : {}), ...(email ? { email } : {}), - ...(orgId ? { orgId } : {}), + ...(credential.orgId ? { orgId: credential.orgId } : {}), }, raw: payload, }; diff --git a/packages/ai/src/usage/kimi.ts b/packages/ai/src/usage/kimi.ts index 17c6a2e6f..10c2db5d5 100644 --- a/packages/ai/src/usage/kimi.ts +++ b/packages/ai/src/usage/kimi.ts @@ -144,15 +144,21 @@ function buildUsageStatus(amount: UsageAmount): UsageStatus { } function toUsageLimit(row: KimiUsageRow, provider: string, index: number, accountId?: string): UsageLimit { - const window: UsageWindow | undefined = - row.window ?? - (row.resetsAt + // Kimi puts `resetTime` on the limit `detail`, not on `window`, so a + // window built from `duration`/`timeUnit` alone carries no resetsAt. + // Fall back to the row-level reset so `omp usage` can render + // "resets in …" for the 5h window too. + const window: UsageWindow | undefined = row.window + ? row.window.resetsAt !== undefined || row.resetsAt === undefined + ? row.window + : { ...row.window, resetsAt: row.resetsAt } + : row.resetsAt ? { id: "default", label: "Usage window", resetsAt: row.resetsAt, } - : undefined); + : undefined; const amount = buildUsageAmount(row); return { diff --git a/packages/ai/src/usage/synthetic.ts b/packages/ai/src/usage/synthetic.ts new file mode 100644 index 000000000..0b932b553 --- /dev/null +++ b/packages/ai/src/usage/synthetic.ts @@ -0,0 +1,180 @@ +import { isRecord } from "@oh-my-pi/pi-utils/type-guards"; +import type { + UsageAmount, + UsageFetchContext, + UsageFetchParams, + UsageLimit, + UsageProvider, + UsageReport, + UsageStatus, + UsageWindow, +} from "../usage"; + +const QUOTAS_URL = "https://api.synthetic.new/v2/quotas"; +const FIVE_HOUR_MS = 5 * 60 * 60 * 1000; +const WEEK_MS = 7 * 24 * 60 * 60 * 1000; + +function parseDollarAmount(value: unknown): number | undefined { + if (typeof value !== "string") return undefined; + const trimmed = value.replace(/^\$/, "").trim(); + const parsed = Number(trimmed); + return Number.isFinite(parsed) ? parsed : undefined; +} + +function parseIsoMs(value: unknown): number | undefined { + if (typeof value !== "string" || !value) return undefined; + const ms = Date.parse(value); + return Number.isFinite(ms) ? ms : undefined; +} + +function buildUsageAmount(args: { + used: number | undefined; + limit: number | undefined; + remaining: number | undefined; + usedFraction: number | undefined; + unit: UsageAmount["unit"]; +}): UsageAmount { + let usedFraction = args.usedFraction; + if (usedFraction === undefined && args.used !== undefined && args.limit !== undefined && args.limit > 0) { + usedFraction = Math.min(args.used / args.limit, 1); + } + const remainingFraction = usedFraction !== undefined ? Math.max(1 - usedFraction, 0) : undefined; + return { + ...(args.used !== undefined ? { used: args.used } : {}), + ...(args.limit !== undefined ? { limit: args.limit } : {}), + ...(args.remaining !== undefined ? { remaining: args.remaining } : {}), + ...(usedFraction !== undefined ? { usedFraction } : {}), + ...(remainingFraction !== undefined ? { remainingFraction } : {}), + unit: args.unit, + }; +} + +function getUsageStatus(usedFraction: number | undefined): UsageStatus | undefined { + if (usedFraction === undefined) return undefined; + if (usedFraction >= 1) return "exhausted"; + if (usedFraction >= 0.9) return "warning"; + return "ok"; +} + +function parseRollingFiveHourLimit(raw: unknown, provider: UsageFetchParams["provider"]): UsageLimit | null { + if (!isRecord(raw)) return null; + const remaining = typeof raw.remaining === "number" ? raw.remaining : undefined; + const max = typeof raw.max === "number" ? raw.max : undefined; + const limited = raw.limited === true; + const nextTickAt = parseIsoMs(raw.nextTickAt); + const tickPercent = typeof raw.tickPercent === "number" ? raw.tickPercent : undefined; + + if (remaining === undefined && max === undefined) return null; + + const used = max !== undefined && remaining !== undefined ? max - remaining : undefined; + const regenPercent = tickPercent !== undefined ? Number((tickPercent * 100).toFixed(2)) : undefined; + const window: UsageWindow = { + id: "5h", + label: regenPercent !== undefined ? `5h · regen ${regenPercent}%/tick` : "5h", + durationMs: FIVE_HOUR_MS, + ...(nextTickAt !== undefined ? { resetsAt: nextTickAt, resetLabel: "tick" } : {}), + }; + const amount = buildUsageAmount({ + used, + limit: max, + remaining, + usedFraction: undefined, + unit: "requests", + }); + const status: UsageStatus = limited ? "exhausted" : (getUsageStatus(amount.usedFraction) ?? "ok"); + return { + id: "synthetic:requests:5h", + label: "Synthetic Requests", + scope: { provider, windowId: "5h", shared: true }, + window, + amount, + status, + }; +} + +function parseWeeklyTokenLimit(raw: unknown, provider: UsageFetchParams["provider"]): UsageLimit | null { + if (!isRecord(raw)) return null; + const remainingCredits = parseDollarAmount(raw.remainingCredits); + const maxCredits = parseDollarAmount(raw.maxCredits); + const percentRemaining = typeof raw.percentRemaining === "number" ? raw.percentRemaining : undefined; + const nextRegenAt = parseIsoMs(raw.nextRegenAt); + + if (remainingCredits === undefined && maxCredits === undefined) return null; + + const usedFraction = + percentRemaining !== undefined ? Math.min(Math.max(1 - percentRemaining / 100, 0), 1) : undefined; + const used = usedFraction !== undefined && maxCredits !== undefined ? usedFraction * maxCredits : undefined; + const nextRegenCredits = parseDollarAmount(raw.nextRegenCredits); + const window: UsageWindow = { + id: "7d", + label: nextRegenCredits !== undefined ? `7d · regen $${nextRegenCredits.toFixed(2)}/tick` : "7d", + durationMs: WEEK_MS, + ...(nextRegenAt !== undefined ? { resetsAt: nextRegenAt, resetLabel: "regen" } : {}), + }; + const amount = buildUsageAmount({ + used, + limit: maxCredits, + remaining: remainingCredits, + usedFraction, + unit: "usd", + }); + return { + id: "synthetic:usd:7d", + label: "Synthetic Credits", + scope: { provider, windowId: "7d", shared: true }, + window, + amount, + status: getUsageStatus(amount.usedFraction), + }; +} + +async function fetchSyntheticUsage(params: UsageFetchParams, ctx: UsageFetchContext): Promise { + if (params.provider !== "synthetic") return null; + const credential = params.credential; + if (credential.type !== "api_key" || !credential.apiKey) return null; + + let payload: unknown = null; + try { + const response = await ctx.fetch(QUOTAS_URL, { + headers: { + Authorization: `Bearer ${credential.apiKey}`, + "Content-Type": "application/json", + }, + signal: params.signal, + }); + if (!response.ok) { + ctx.logger?.warn("Synthetic usage fetch failed", { status: response.status, statusText: response.statusText }); + return null; + } + payload = await response.json(); + } catch (error) { + ctx.logger?.warn("Synthetic usage fetch error", { error: String(error) }); + return null; + } + + if (!isRecord(payload)) return null; + + const limits: UsageLimit[] = []; + + const fiveHour = parseRollingFiveHourLimit(payload.rollingFiveHourLimit, params.provider); + if (fiveHour) limits.push(fiveHour); + + const weekly = parseWeeklyTokenLimit(payload.weeklyTokenLimit, params.provider); + if (weekly) limits.push(weekly); + + if (limits.length === 0) return null; + + return { + provider: params.provider, + fetchedAt: Date.now(), + limits, + metadata: { endpoint: QUOTAS_URL }, + raw: payload, + }; +} + +export const syntheticUsageProvider: UsageProvider = { + id: "synthetic", + fetchUsage: fetchSyntheticUsage, + supports: params => params.provider === "synthetic" && params.credential.type === "api_key", +}; diff --git a/packages/ai/src/usage/xai-oauth.ts b/packages/ai/src/usage/xai-oauth.ts new file mode 100644 index 000000000..98a6bf52e --- /dev/null +++ b/packages/ai/src/usage/xai-oauth.ts @@ -0,0 +1,256 @@ +/** + * SuperGrok (`xai-oauth`) subscription usage provider. + * + * Reads weekly credit and product utilization from the Grok CLI billing + * endpoint. Only OAuth access credentials are accepted; paid API keys are a + * separate product and must never be sent here. + */ + +import { + buildXAICliBillingUrl, + extractXAIAccessTokenSubject, + fetchXAIOAuthIdentity, + getXAICliBillingHeaders, +} from "../registry/oauth/xai-oauth"; +import type { + UsageAmount, + UsageFetchContext, + UsageFetchParams, + UsageLimit, + UsageProvider, + UsageReport, + UsageStatus, + UsageWindow, +} from "../usage"; +import { isRecord } from "../utils"; +import { toNumber } from "./shared"; + +const PROVIDER_ID = "xai-oauth"; +const WEEK_MS = 7 * 24 * 60 * 60 * 1000; + +interface XaiBillingPeriod { + start: string; + end: string; + type: string; +} + +interface XaiProductUsage { + product: string; + usagePercent: number; +} + +interface XaiBillingConfig { + currentPeriod: XaiBillingPeriod; + creditUsagePercent: number; + productUsage: XaiProductUsage[]; + onDemandCap?: number; + onDemandUsed?: number; +} + +function parseIsoMs(value: string): number | undefined { + const parsed = Date.parse(value); + return Number.isFinite(parsed) ? parsed : undefined; +} + +function parsePercent(value: unknown): number | undefined { + const percent = toNumber(value); + return percent !== undefined && percent >= 0 && percent <= 100 ? percent : undefined; +} + +function parseOnDemandAmount(value: unknown): number | undefined { + if (!isRecord(value)) return undefined; + const amount = toNumber(value.val); + return amount !== undefined && amount >= 0 ? amount : undefined; +} + +function buildPercentAmount(usagePercent: number): UsageAmount { + const usedFraction = usagePercent / 100; + return { + used: usagePercent, + limit: 100, + remaining: 100 - usagePercent, + usedFraction, + remainingFraction: 1 - usedFraction, + unit: "percent", + }; +} + +function buildUsageStatus(usedFraction: number): UsageStatus { + if (usedFraction >= 1) return "exhausted"; + if (usedFraction >= 0.9) return "warning"; + return "ok"; +} + +function slugifyProduct(product: string): string { + return product + .trim() + .toLowerCase() + .replace(/[^a-z0-9]+/g, "-") + .replace(/^-+|-+$/g, ""); +} + +function buildPeriodWindow(period: XaiBillingPeriod): UsageWindow { + return { + id: "1w", + label: "Weekly", + durationMs: WEEK_MS, + resetsAt: parseIsoMs(period.end), + }; +} + +function parseBillingConfig(payload: unknown): XaiBillingConfig | null { + if (!isRecord(payload) || !isRecord(payload.config)) return null; + const raw = payload.config; + if (!isRecord(raw.currentPeriod)) return null; + + const start = typeof raw.currentPeriod.start === "string" ? parseIsoMs(raw.currentPeriod.start) : undefined; + const end = typeof raw.currentPeriod.end === "string" ? parseIsoMs(raw.currentPeriod.end) : undefined; + const type = typeof raw.currentPeriod.type === "string" ? raw.currentPeriod.type : ""; + // Keep recently-ended weekly windows so /usage still renders across period + // rollover while the billing API is mid-refresh. Reject only inverted ranges + // and non-weekly period types. + if (start === undefined || end === undefined || end <= start || !type.toUpperCase().includes("WEEK")) { + return null; + } + + const creditUsagePercent = parsePercent(raw.creditUsagePercent); + if (creditUsagePercent === undefined) return null; + + const productUsage: XaiProductUsage[] = []; + if (raw.productUsage !== undefined) { + if (!Array.isArray(raw.productUsage)) return null; + for (const item of raw.productUsage) { + if (!isRecord(item)) continue; + const product = typeof item.product === "string" ? item.product.trim() : ""; + const usagePercent = parsePercent(item.usagePercent); + if (!product || usagePercent === undefined) continue; + productUsage.push({ product, usagePercent }); + } + } + + return { + currentPeriod: { + start: raw.currentPeriod.start as string, + end: raw.currentPeriod.end as string, + type, + }, + creditUsagePercent, + productUsage, + onDemandCap: parseOnDemandAmount(raw.onDemandCap), + onDemandUsed: parseOnDemandAmount(raw.onDemandUsed), + }; +} + +function buildLimits(config: XaiBillingConfig, accountId: string | undefined): UsageLimit[] { + const window = buildPeriodWindow(config.currentPeriod); + const scope = { + provider: PROVIDER_ID, + ...(accountId ? { accountId } : {}), + windowId: window.id, + shared: true as const, + }; + const overall = buildPercentAmount(config.creditUsagePercent); + const limits: UsageLimit[] = [ + { + id: `${PROVIDER_ID}:credits:1w`, + label: "SuperGrok Weekly Credits", + scope, + window, + amount: overall, + status: buildUsageStatus(overall.usedFraction ?? 0), + }, + ]; + + for (const item of config.productUsage) { + const amount = buildPercentAmount(item.usagePercent); + const slug = slugifyProduct(item.product); + if (!slug) continue; + limits.push({ + id: `${PROVIDER_ID}:product:${slug}:1w`, + label: `${item.product === "GrokBuild" ? "Grok Build" : item.product === "Api" ? "API" : item.product} (Weekly)`, + scope, + window, + amount, + status: buildUsageStatus(amount.usedFraction ?? 0), + }); + } + if (config.onDemandCap !== undefined && config.onDemandCap > 0 && config.onDemandUsed !== undefined) { + const usedFraction = Math.min(config.onDemandUsed / config.onDemandCap, 1); + limits.push({ + id: `${PROVIDER_ID}:on-demand`, + label: "On-demand", + scope: { + provider: PROVIDER_ID, + ...(accountId ? { accountId } : {}), + shared: true, + }, + amount: { + used: config.onDemandUsed, + limit: config.onDemandCap, + remaining: Math.max(0, config.onDemandCap - config.onDemandUsed), + usedFraction, + remainingFraction: 1 - usedFraction, + unit: "unknown", + }, + status: buildUsageStatus(usedFraction), + }); + } + + return limits; +} + +export const xaiOauthUsageProvider: UsageProvider = { + id: PROVIDER_ID, + + supports(params: UsageFetchParams): boolean { + return params.provider === PROVIDER_ID && params.credential.type === "oauth" && !!params.credential.accessToken; + }, + + async fetchUsage(params: UsageFetchParams, ctx: UsageFetchContext): Promise { + if (params.provider !== PROVIDER_ID || params.credential.type !== "oauth") return null; + const accessToken = params.credential.accessToken?.trim(); + if (!accessToken) return null; + if (params.credential.expiresAt !== undefined && params.credential.expiresAt <= Date.now()) return null; + + let accountId = params.credential.accountId?.trim() || extractXAIAccessTokenSubject(accessToken); + let email = params.credential.email?.trim().toLowerCase(); + if (!email) { + try { + const identity = await fetchXAIOAuthIdentity(accessToken, ctx.fetch, params.signal); + email = identity?.email?.trim().toLowerCase() || undefined; + accountId ??= identity?.accountId?.trim() || undefined; + } catch { + // Identity enrichment is best effort; billing remains authoritative. + } + } + + const url = buildXAICliBillingUrl(); + let payload: unknown; + try { + const response = await ctx.fetch(url, { + headers: getXAICliBillingHeaders({ accessToken }), + redirect: "error", + signal: params.signal, + }); + if (!response.ok) return null; + payload = await response.json(); + } catch { + return null; + } + + const config = parseBillingConfig(payload); + if (!config) return null; + return { + provider: PROVIDER_ID, + fetchedAt: Date.now(), + limits: buildLimits(config, accountId), + metadata: { + endpoint: url, + source: "cli-chat-proxy.grok.com/v1/billing", + ...(accountId ? { accountId } : {}), + ...(email ? { email } : {}), + }, + raw: payload, + }; + }, +}; diff --git a/packages/ai/src/utils.ts b/packages/ai/src/utils.ts index 0445cd3ac..ba747da38 100644 --- a/packages/ai/src/utils.ts +++ b/packages/ai/src/utils.ts @@ -1,5 +1,6 @@ import { $env } from "@oh-my-pi/pi-utils"; import type { ResponseInput, ResponseInputItem } from "./providers/openai-responses-wire"; +import { redactSensitiveCredentials } from "./providers/transform-messages"; import type { CacheRetention, OpenAIResponsesHistoryPayload, ProviderPayload } from "./types"; type OpenAIResponsesReplayItem = ResponseInput[number]; @@ -9,7 +10,9 @@ export { isRecord } from "@oh-my-pi/pi-utils"; export function normalizeSystemPrompts(systemPrompt: readonly string[] | string | undefined | null): string[] { if (systemPrompt === undefined || systemPrompt === null) return []; const prompts = Array.isArray(systemPrompt) ? systemPrompt : typeof systemPrompt === "string" ? [systemPrompt] : []; - return prompts.map(prompt => prompt.toWellFormed()).filter(prompt => prompt.trim().length > 0); + return prompts + .map(prompt => redactSensitiveCredentials(prompt.toWellFormed())) + .filter(prompt => prompt.trim().length > 0); } export function normalizeToolCallId(id: string): string { @@ -284,11 +287,16 @@ export function getOpenAIResponsesHistoryItems( } /** - * Resolve cache retention preference. - * Defaults to "short" and uses PI_CACHE_RETENTION for backward compatibility. + * Resolve cache retention preference: explicit request option first, then the + * `PI_CACHE_RETENTION` env override (`long` | `short` | `none`), then the + * provider-supplied fallback. */ -export function resolveCacheRetention(cacheRetention?: CacheRetention): CacheRetention { +export function resolveCacheRetention( + cacheRetention?: CacheRetention, + fallback: CacheRetention = "short", +): CacheRetention { if (cacheRetention) return cacheRetention; - if ($env.PI_CACHE_RETENTION === "long") return "long"; - return "short"; + const env = $env.PI_CACHE_RETENTION; + if (env === "long" || env === "short" || env === "none") return env; + return fallback; } diff --git a/packages/ai/src/utils/empty-completion-retry.ts b/packages/ai/src/utils/empty-completion-retry.ts index 6ace44c1f..3af9337e5 100644 --- a/packages/ai/src/utils/empty-completion-retry.ts +++ b/packages/ai/src/utils/empty-completion-retry.ts @@ -27,12 +27,13 @@ export const EMPTY_COMPLETION_BASE_DELAY_MS = 500; const NON_WHITESPACE_RE = /\S/; /** - * Whether a completed assistant message carries content worth delivering: a tool - * call or any non-whitespace text. An empty/whitespace-only message — or one - * that only ever produced thinking — is the "empty response" failure. + * Whether a completed assistant message carries content worth delivering: an + * image, tool call, or any non-whitespace text. An empty/whitespace-only message + * — or one that only ever produced thinking — is the "empty response" failure. */ export function hasVisibleAssistantContent(message: AssistantMessage): boolean { for (const block of message.content) { + if (block.type === "image") return true; if (block.type === "toolCall") return true; if (block.type === "text" && NON_WHITESPACE_RE.test(block.text)) return true; } @@ -49,6 +50,8 @@ function isMeaningfulCompletionEvent(event: AssistantMessageEvent): boolean { case "text_end": case "thinking_end": return event.content.length > 0; + case "image_end": + return true; case "toolcall_start": case "toolcall_end": return true; diff --git a/packages/ai/src/utils/event-stream.ts b/packages/ai/src/utils/event-stream.ts index 6d1255e39..6cdcc197b 100644 --- a/packages/ai/src/utils/event-stream.ts +++ b/packages/ai/src/utils/event-stream.ts @@ -195,3 +195,8 @@ export class AssistantMessageEventStream extends EventStream(); + /** Projected native thinking blocks, keyed by the inner stream's `contentIndex`. */ + #thinkingBlocks = new Map(); + /** Native thinking blocks whose projected `thinking_end` awaits the source end event. */ + #pendingThinkingEnds = new Set(); constructor(out: AssistantMessageEventStream, seed: AssistantMessage) { this.#out = out; @@ -154,15 +175,51 @@ class LeakedThinkingProjector { this.#apply(this.#healer.feedEvents(delta), this.#lastTextSignature); } - /** Forward a native thinking delta, preserving its signature. */ - thinking(delta: string, signature: string | undefined): void { - const index = this.#openThinking(); + /** Forward a native thinking delta, preserving its source block identity and signature. */ + thinking(srcIndex: number, delta: string, signature: string | undefined): void { + let index = this.#thinkingBlocks.get(srcIndex); + if (index === undefined) { + if (this.#thinking && this.#pendingThinkingEnds.has(this.#thinking.index)) this.#closeThinking(); + index = this.#openThinking(); + this.#thinkingBlocks.set(srcIndex, index); + this.#pendingThinkingEnds.add(index); + } const block = this.#partial.content[index] as ThinkingContent; block.thinking += delta; if (signature !== undefined) block.thinkingSignature = signature; this.#out.push({ type: "thinking_delta", contentIndex: index, delta, partial: this.#partial }); } + /** + * Finalize a native thinking block by source identity. Its projected end is + * deferred until this event so stream consumers observe the completed + * signature before the block closes, even when later blocks started first. + */ + thinkingEnd(srcIndex: number, signature: string | undefined): void { + const index = this.#thinkingBlocks.get(srcIndex); + if (index === undefined) return; + if (signature !== undefined) { + (this.#partial.content[index] as ThinkingContent).thinkingSignature = signature; + } + if (!this.#pendingThinkingEnds.delete(index)) return; + if (this.#thinking?.index === index) this.#thinking = undefined; + this.#emitThinkingEnd(index); + } + + /** Forward a completed native image after releasing held text. */ + image(content: ImageContent): void { + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + this.#partial.content.push(content); + this.#out.push({ + type: "image_end", + contentIndex: this.#partial.content.length - 1, + content, + partial: this.#partial, + }); + } + /** Forward a native tool call's start, releasing any held-back text first. */ toolStart(srcIndex: number, source: StreamingToolCall | undefined): void { if (!source) return; @@ -216,6 +273,10 @@ class LeakedThinkingProjector { * flush held-back fragments, close open blocks, and return the healed content. */ finish(message: AssistantMessage): AssistantMessage["content"] { + for (const [srcIndex] of this.#thinkingBlocks) { + const block = message.content[srcIndex]; + this.thinkingEnd(srcIndex, block?.type === "thinking" ? block.thinkingSignature : undefined); + } let fullText = ""; let tailSignature: string | undefined; for (const block of message.content) { @@ -286,13 +347,19 @@ class LeakedThinkingProjector { #closeThinking(): void { if (!this.#thinking) return; - const block = this.#partial.content[this.#thinking.index] as ThinkingContent; + const index = this.#thinking.index; + this.#thinking = undefined; + if (this.#pendingThinkingEnds.has(index)) return; + this.#emitThinkingEnd(index); + } + + #emitThinkingEnd(index: number): void { + const block = this.#partial.content[index] as ThinkingContent; this.#out.push({ type: "thinking_end", - contentIndex: this.#thinking.index, + contentIndex: index, content: block.thinking, partial: this.#partial, }); - this.#thinking = undefined; } } diff --git a/packages/ai/src/utils/proxy.ts b/packages/ai/src/utils/proxy.ts index 00e5e0f50..33fb0b12b 100644 --- a/packages/ai/src/utils/proxy.ts +++ b/packages/ai/src/utils/proxy.ts @@ -61,7 +61,7 @@ export function shouldBypassProxy(urlObj: URL): boolean { .map(r => r.trim()) .filter(Boolean); const targetHost = urlObj.hostname.toLowerCase(); - const targetPort = urlObj.port || (urlObj.protocol === "https:" ? "443" : "80"); + const targetPort = urlObj.port || (urlObj.protocol === "https:" || urlObj.protocol === "wss:" ? "443" : "80"); for (const rule of rules) { if (rule === "*") { diff --git a/packages/ai/src/utils/schema/CONSTRAINTS.md b/packages/ai/src/utils/schema/CONSTRAINTS.md index ee4a9e7a6..a06f47d66 100644 --- a/packages/ai/src/utils/schema/CONSTRAINTS.md +++ b/packages/ai/src/utils/schema/CONSTRAINTS.md @@ -69,6 +69,7 @@ Schemas sent on the Google JSON Schema path MUST follow: - `minItems`, `maxItems`, `minLength`, `maxLength` - `minimum`, `maximum`, `exclusiveMinimum`, `exclusiveMaximum` - `pattern`, `format` + - `dependencies`, `dependentSchemas`, `dependentRequired` - Important: keys inside a `properties` object are treated as property names and MUST NOT be stripped by keyword match. - Human-meaningful stripped keys (`pattern`, `format`, min/max constraints, `default`, `examples`, etc.) are appended to the sibling `description` as an Anthropic-style spill block: `{pattern: "^foo$", minimum: 0}`. Structural/meta keys such as `$ref`, `$defs`, and `additionalProperties` are not spilled. diff --git a/packages/ai/src/utils/schema/fields.ts b/packages/ai/src/utils/schema/fields.ts index b25006248..95edbb93e 100644 --- a/packages/ai/src/utils/schema/fields.ts +++ b/packages/ai/src/utils/schema/fields.ts @@ -37,6 +37,9 @@ export const UNSUPPORTED_SCHEMA_FIELDS: Record = { multipleOf: true, pattern: true, format: true, + dependencies: true, + dependentSchemas: true, + dependentRequired: true, }; /** diff --git a/packages/ai/src/utils/schema/normalize.ts b/packages/ai/src/utils/schema/normalize.ts index e651a1b07..7cd2a6d37 100644 --- a/packages/ai/src/utils/schema/normalize.ts +++ b/packages/ai/src/utils/schema/normalize.ts @@ -26,9 +26,14 @@ import { enter, epochNext, exit, once, stamp } from "./stamps"; import { isJsonObject, isJsonObjectEmpty, type JsonObject } from "./types"; import { decontaminateZodInstance } from "./zod-decontaminate"; -export type ResidualSchemaIncompatibility = "type-array" | "type-null" | "nullable" | "combiners"; +export type ResidualSchemaIncompatibility = "type-array" | "type-null" | "nullable" | "combiners" | "not"; export interface NormalizeSchemaOptions { + /** + * Coerce boolean subschemas to object forms. `standard` preserves `false` + * with `not`; `permissive` uses `{}` when the provider cannot express it. + */ + coerceBooleanSubschemas?: "standard" | "permissive"; unsupportedFields: (key: string) => boolean; normalizeFieldNames: boolean; collapseNullFields: boolean; @@ -55,7 +60,14 @@ export interface NormalizeSchemaOptions { } interface NormalizeSchemaWalkOptions extends NormalizeSchemaOptions { - insideProperties: boolean; + insideSchemaMap: boolean; + /** + * True when the value currently being walked occupies a JSON Schema + * *subschema* slot (root, combiner branch, `items`, a property value, …). + * Only then is a bare `true`/`false` a boolean subschema to coerce; in a + * keyword slot (`nullable`, `enum` entries, `additionalProperties`) it stays. + */ + booleanIsSubschema: boolean; } interface ResidualIncompatibilityChecks { @@ -63,6 +75,7 @@ interface ResidualIncompatibilityChecks { typeNull: boolean; nullable: boolean; combiners: boolean; + not: boolean; } const SNAKE_TO_CAMEL_RENAMES = new Map([ @@ -75,6 +88,42 @@ const SNAKE_TO_CAMEL_RENAMES = new Map([ const JSON_SCHEMA_COMBINERS = ["anyOf", "oneOf"] as const; const CCA_FORBIDDEN_COMBINERS = new Set(["anyOf", "oneOf", "allOf"]); +/** + * Keywords whose value is a single subschema (draft 2020-12). A bare `true` / + * `false` in one of these slots is a boolean subschema to coerce (issue #5604). + */ +const SUBSCHEMA_VALUE_KEYS: Record = { + items: true, + additionalItems: true, + unevaluatedItems: true, + not: true, + if: true, + // biome-ignore lint/suspicious/noThenProperty: JSON Schema keyword + then: true, + else: true, + contains: true, + propertyNames: true, + contentSchema: true, +}; + +/** Keywords whose value is an array of subschemas. */ +const SUBSCHEMA_ARRAY_KEYS: Record = { + anyOf: true, + oneOf: true, + allOf: true, + prefixItems: true, +}; + +/** Keywords whose object value maps arbitrary names to subschemas. */ +const SUBSCHEMA_MAP_KEYS: Record = { + properties: true, + patternProperties: true, + dependencies: true, + dependentSchemas: true, + $defs: true, + definitions: true, +}; + const CLOUD_CODE_ASSIST_CLAUDE_FALLBACK_SCHEMA = { type: "object", properties: {}, @@ -236,6 +285,15 @@ function normalizeSchemaNode(value: unknown, options: NormalizeSchemaWalkOptions exit(value); } } + if (typeof value === "boolean") { + // A bare boolean is a JSON Schema subschema only in a subschema slot. + // Some provider wires have no boolean-schema representation: `true` + // becomes `{}`; `false` uses `not` when supported, or the permissive + // `{}` fallback when the provider cannot express an impossible schema. + const mode = options.coerceBooleanSubschemas; + if (!mode || !options.booleanIsSubschema) return value; + return value || mode === "permissive" ? {} : { not: {} }; + } if (!isJsonObject(value)) { return value; } @@ -250,8 +308,8 @@ function normalizeSchemaNode(value: unknown, options: NormalizeSchemaWalkOptions } function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWalkOptions): unknown { - let obj = options.normalizeFieldNames && !options.insideProperties ? applySnakeCaseRenames(value) : value; - if (options.collapseNullFields && !options.insideProperties) { + let obj = options.normalizeFieldNames && !options.insideSchemaMap ? applySnakeCaseRenames(value) : value; + if (options.collapseNullFields && !options.insideSchemaMap) { obj = preHandleNullFields(obj); } const result: JsonObject = {}; @@ -298,14 +356,18 @@ function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWa for (const key in obj) { if (!Object.hasOwn(obj, key) || key === combiner || outHasOwn(result, key)) continue; const entry = obj[key]; - if (!options.insideProperties && options.unsupportedFields(key)) { + if (!options.insideSchemaMap && options.unsupportedFields(key)) { spill = pushStrippedDescriptionEntry(spill, key, entry, options); continue; } if (options.stripNullableKeyword && key === "nullable") continue; result[key] = normalizeSchemaNode(entry, { ...options, - insideProperties: !options.insideProperties && key === "properties", + insideSchemaMap: !options.insideSchemaMap && Object.hasOwn(SUBSCHEMA_MAP_KEYS, key), + booleanIsSubschema: + options.insideSchemaMap || + Object.hasOwn(SUBSCHEMA_VALUE_KEYS, key) || + Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, key), }); } applyDescriptionSpill(result, spill, options); @@ -316,7 +378,7 @@ function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWa for (const key in obj) { if (!Object.hasOwn(obj, key)) continue; const entry = obj[key]; - if (!options.insideProperties && options.unsupportedFields(key)) { + if (!options.insideSchemaMap && options.unsupportedFields(key)) { spill = pushStrippedDescriptionEntry(spill, key, entry, options); continue; } @@ -327,7 +389,11 @@ function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWa } result[key] = normalizeSchemaNode(entry, { ...options, - insideProperties: !options.insideProperties && key === "properties", + insideSchemaMap: !options.insideSchemaMap && Object.hasOwn(SUBSCHEMA_MAP_KEYS, key), + booleanIsSubschema: + options.insideSchemaMap || + Object.hasOwn(SUBSCHEMA_VALUE_KEYS, key) || + Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, key), }); } @@ -835,6 +901,7 @@ function createResidualIncompatibilityChecks( typeNull: false, nullable: false, combiners: false, + not: false, }; for (const check of checks) { switch (check) { @@ -847,6 +914,9 @@ function createResidualIncompatibilityChecks( case "nullable": result.nullable = true; break; + case "not": + result.not = true; + break; case "combiners": result.combiners = true; break; @@ -859,10 +929,11 @@ function hasResidualSchemaIncompatibilities( value: unknown, checks: ResidualIncompatibilityChecks, epoch: number = epochNext(), + insideSchemaMap = false, ): boolean { if (Array.isArray(value)) { if (!once(value, epoch)) return false; - return value.some(entry => hasResidualSchemaIncompatibilities(entry, checks, epoch)); + return value.some(entry => hasResidualSchemaIncompatibilities(entry, checks, epoch, insideSchemaMap)); } if (!isJsonObject(value)) { return false; @@ -874,14 +945,22 @@ function hasResidualSchemaIncompatibilities( if (checks.typeArray && Array.isArray(value.type)) return true; if (checks.typeNull && value.type === "null") return true; if (checks.nullable && Object.hasOwn(value, "nullable")) return true; - if (checks.combiners) { + if (!insideSchemaMap && checks.not && Object.hasOwn(value, "not")) return true; + if (!insideSchemaMap && checks.combiners) { for (const combiner of CCA_FORBIDDEN_COMBINERS) { if (Array.isArray(value[combiner])) return true; } } for (const k in value) { if (!Object.hasOwn(value, k)) continue; - if (hasResidualSchemaIncompatibilities(value[k], checks, epoch)) { + if ( + hasResidualSchemaIncompatibilities( + value[k], + checks, + epoch, + !insideSchemaMap && Object.hasOwn(SUBSCHEMA_MAP_KEYS, k), + ) + ) { return true; } } @@ -894,7 +973,8 @@ export function normalizeSchema(value: unknown, options: NormalizeSchemaOptions) const dereferenced = dereferenceJsonSchema(upgraded); let normalized = normalizeSchemaNode(dereferenced, { ...options, - insideProperties: false, + insideSchemaMap: false, + booleanIsSubschema: true, }); if (options.stripResidualCombinersFixpoint) { normalized = stripResidualCombiners(normalized); @@ -916,6 +996,7 @@ export function normalizeSchema(value: unknown, options: NormalizeSchemaOptions) export function normalizeSchemaForGoogle(value: unknown): unknown { return normalizeSchema(value, { + coerceBooleanSubschemas: "standard", unsupportedFields: isGoogleUnsupportedSchemaField, normalizeFieldNames: true, collapseNullFields: true, @@ -937,6 +1018,7 @@ export function normalizeSchemaForGoogle(value: unknown): unknown { export function normalizeSchemaForCCA(value: unknown): unknown { return normalizeSchema(value, { + coerceBooleanSubschemas: "standard", unsupportedFields: isGoogleUnsupportedSchemaField, normalizeFieldNames: true, collapseNullFields: false, @@ -953,7 +1035,7 @@ export function normalizeSchemaForCCA(value: unknown): unknown { inferTypeForBareEnum: true, dropNonScalarEnum: false, foldOneOfIntoAnyOf: false, - rejectResidualIncompatibilities: ["type-array", "type-null", "nullable", "combiners"], + rejectResidualIncompatibilities: ["type-array", "type-null", "nullable", "combiners", "not"], validateAndFallback: { fallback: CLOUD_CODE_ASSIST_CLAUDE_FALLBACK_SCHEMA }, }); } @@ -1001,6 +1083,9 @@ export function normalizeSchemaForMCP(value: unknown): unknown { * `default` and `description` are MFJS Meta Data fields and are preserved. * - `additionalProperties` (boolean or schema) and `type: "null"` (incl. * inside `anyOf`) are kept. + * - Boolean subschemas are object-coerced; MFJS has no exact `false` schema, + * so both values become the permissive empty schema while local tool + * validation remains authoritative. * * Out of scope (absent from the built-in tool surface, spec-ambiguous to * rewrite blindly): `allOf` intersection merging, external/recursive `$ref`, @@ -1008,6 +1093,7 @@ export function normalizeSchemaForMCP(value: unknown): unknown { */ export function normalizeSchemaForMoonshot(value: unknown): unknown { return normalizeSchema(value, { + coerceBooleanSubschemas: "permissive", unsupportedFields: isMoonshotUnsupportedSchemaField, normalizeFieldNames: false, collapseNullFields: false, @@ -1031,15 +1117,6 @@ export function normalizeSchemaForMoonshot(value: unknown): unknown { // Ollama — Go schema parser compatibility // --------------------------------------------------------------------------- -const OLLAMA_SCHEMA_ARRAY_KEYS = new Set(["anyOf", "oneOf", "allOf", "prefixItems"]); -const OLLAMA_SCHEMA_MAP_KEYS = new Set([ - "properties", - "patternProperties", - "dependencies", - "dependentSchemas", - "$defs", - "definitions", -]); const OLLAMA_SCHEMA_VALUE_KEYS = new Set([ "items", "additionalItems", @@ -1056,18 +1133,19 @@ const OLLAMA_SCHEMA_VALUE_KEYS = new Set([ ]); /** - * Widened stand-in for a `true` / `{}` subschema in an Ollama-bound tool. + * Widened stand-in for a `true` / `{}` open subschema on a tool bound for a + * backend whose wire cannot encode a bare boolean subschema. * * `toolWireSchema()` normalizes empty schemas to boolean `true` upstream so - * grammar-constrained samplers (llama.cpp, etc.) don't treat `{}` as - * "generate an empty object" (issue #1179). Ollama's Go tool parser can't - * unmarshal a boolean into its object-shaped `Schema` struct, so this - * sanitizer replaces every open subschema with an explicit union of every - * primitive JSON type. Both invariants survive: the wire has no boolean - * subschema (Go accepts it), and llama.cpp's grammar sees a real value - * union rather than a closed empty object. + * grammar-constrained samplers don't treat `{}` as "generate an empty object" + * (issue #1179). Two backends then choke on the bare boolean: Ollama's Go tool + * parser can't unmarshal it into its object-shaped `Schema` struct, and + * llama.cpp's JSON-schema→GBNF converter has no case for a boolean schema + * (issue #5914). Both sanitizers replace the open subschema with an explicit + * union of every primitive JSON type — the wire has no boolean subschema, and + * a grammar sampler sees a real value union rather than a closed empty object. */ -const OLLAMA_OPEN_SUBSCHEMA_WIDENING = Object.freeze({ +const OPEN_SUBSCHEMA_WIDENING = Object.freeze({ anyOf: [ { type: "string" }, { type: "number" }, @@ -1084,8 +1162,8 @@ const OLLAMA_OPEN_SUBSCHEMA_WIDENING = Object.freeze({ */ export function sanitizeSchemaForOllama(schema: JsonObject): JsonObject { const normalizeNode = (value: unknown): unknown => { - if (value === true) return OLLAMA_OPEN_SUBSCHEMA_WIDENING; - if (value === false) return { not: OLLAMA_OPEN_SUBSCHEMA_WIDENING }; + if (value === true) return OPEN_SUBSCHEMA_WIDENING; + if (value === false) return { not: OPEN_SUBSCHEMA_WIDENING }; if (!isJsonObject(value)) { if (!Array.isArray(value)) return value; let changed = false; @@ -1121,7 +1199,7 @@ export function sanitizeSchemaForOllama(schema: JsonObject): JsonObject { } let next = child; - if (OLLAMA_SCHEMA_MAP_KEYS.has(key) && isJsonObject(child)) { + if (Object.hasOwn(SUBSCHEMA_MAP_KEYS, key) && isJsonObject(child)) { let mapChanged = false; const mapOutput: JsonObject = {}; for (const childKey in child) { @@ -1132,7 +1210,7 @@ export function sanitizeSchemaForOllama(schema: JsonObject): JsonObject { mapOutput[childKey] = normalizedChild; } next = mapChanged ? mapOutput : child; - } else if (OLLAMA_SCHEMA_ARRAY_KEYS.has(key) && Array.isArray(child)) { + } else if (Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, key) && Array.isArray(child)) { let arrayChanged = false; const arrayOutput = child.map(item => { const normalizedItem = normalizeNode(item); @@ -1158,6 +1236,97 @@ export function sanitizeSchemaForOllama(schema: JsonObject): JsonObject { return normalizeNode(schema) as JsonObject; } +/** + * Schema-valued keywords whose bare boolean value must be widened for a + * grammar-constrained backend. Excludes `additionalProperties` and + * `unevaluatedProperties`: llama.cpp's `_build_object_rule` reads their boolean + * form as meaningful closed/open-object semantics, and `additionalProperties: + * false` is exactly what `toolWireSchema` emits to pin a strict object shape. + */ +const GRAMMAR_SCHEMA_VALUE_KEYS: Record = { + items: true, + additionalItems: true, + contains: true, + contentSchema: true, + propertyNames: true, + if: true, + // biome-ignore lint/suspicious/noThenProperty: JSON Schema keyword + then: true, + else: true, + not: true, + unevaluatedItems: true, +}; + +/** + * Rewrites the one JSON Schema form that grammar-constrained OpenAI-compatible + * backends (llama.cpp, LM Studio, vLLM) cannot compile to GBNF: a bare boolean + * subschema. `toolWireSchema` normalizes `{}` open subschemas to boolean `true` + * (issue #1179); llama.cpp's `json-schema-to-grammar.cpp` `visit()` has no case + * for a boolean schema and throws `Unrecognized schema: true` → HTTP 400 before + * the model is consulted (issue #5914). + * + * Narrower than {@link sanitizeSchemaForOllama}: only genuine subschema slots + * are widened. Boolean `additionalProperties`/`unevaluatedProperties` stay + * intact because the converter reads those as closed/open-object grammar + * semantics, and dropping `additionalProperties: false` would silently reopen + * every declared object. + */ +export function sanitizeSchemaForGrammar(schema: JsonObject): JsonObject { + const normalizeNode = (value: unknown, isSubschema: boolean): unknown => { + if (value === true) return isSubschema ? OPEN_SUBSCHEMA_WIDENING : value; + if (value === false) return isSubschema ? { not: OPEN_SUBSCHEMA_WIDENING } : value; + if (Array.isArray(value)) { + let changed = false; + const output = value.map(item => { + const next = normalizeNode(item, isSubschema); + if (next !== item) changed = true; + return next; + }); + return changed ? output : value; + } + if (!isJsonObject(value)) return value; + + let changed = false; + const output: JsonObject = {}; + for (const key in value) { + if (!Object.hasOwn(value, key)) continue; + const child = value[key]; + let next = child; + if (Object.hasOwn(SUBSCHEMA_MAP_KEYS, key) && isJsonObject(child)) { + let mapChanged = false; + const mapOutput: JsonObject = {}; + for (const childKey in child) { + if (!Object.hasOwn(child, childKey)) continue; + const mapChild = child[childKey]; + const normalizedChild = normalizeNode(mapChild, true); + if (normalizedChild !== mapChild) mapChanged = true; + mapOutput[childKey] = normalizedChild; + } + next = mapChanged ? mapOutput : child; + } else if (Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, key) && Array.isArray(child)) { + let arrayChanged = false; + const arrayOutput = child.map(item => { + const normalizedItem = normalizeNode(item, true); + if (normalizedItem !== item) arrayChanged = true; + return normalizedItem; + }); + next = arrayChanged ? arrayOutput : child; + } else if (Object.hasOwn(GRAMMAR_SCHEMA_VALUE_KEYS, key)) { + next = normalizeNode(child, true); + } else if ((key === "additionalProperties" || key === "unevaluatedProperties") && typeof child !== "boolean") { + // Boolean form is meaningful closed/open-object grammar semantics and + // stays intact; the object form is a genuine subschema whose interior + // may still hold bare booleans emitted by `toolWireSchema`. + next = normalizeNode(child, true); + } + if (next !== child) changed = true; + output[key] = next; + } + return changed ? output : value; + }; + return normalizeNode(schema, true) as JsonObject; +} + // --------------------------------------------------------------------------- // OpenAI Responses — schema-valued normalization // --------------------------------------------------------------------------- diff --git a/packages/ai/test/alibaba-endpoint-selection.test.ts b/packages/ai/test/alibaba-endpoint-selection.test.ts index 725734419..30cd8e266 100644 --- a/packages/ai/test/alibaba-endpoint-selection.test.ts +++ b/packages/ai/test/alibaba-endpoint-selection.test.ts @@ -7,13 +7,16 @@ import type { OAuthController } from "../src/registry/oauth/types"; describe("alibaba-coding-plan endpoint selection", () => { let validateSpy: Mock; + let validateModelsSpy: Mock; beforeEach(() => { validateSpy = spyOn(apiKeyValidation, "validateOpenAICompatibleApiKey").mockResolvedValue(undefined); + validateModelsSpy = spyOn(apiKeyValidation, "validateApiKeyAgainstModelsEndpoint").mockResolvedValue(undefined); }); afterEach(() => { validateSpy.mockRestore(); + validateModelsSpy.mockRestore(); }); it("option 1 uses international endpoint and auth URL", async () => { @@ -79,6 +82,7 @@ describe("alibaba-coding-plan endpoint selection", () => { }); it("option 3 prompts for custom URL and uses it", async () => { + validateSpy.mockRejectedValue(new Error("404 model_not_found")); let capturedAuth: { url: string; instructions?: string } | undefined; const options: OAuthController = { onAuth: info => { @@ -100,11 +104,10 @@ describe("alibaba-coding-plan endpoint selection", () => { expect(result.enterpriseUrl).toBe("https://my-proxy.com/v1"); expect(capturedAuth?.url).toBe("https://modelstudio.console.alibabacloud.com/"); - expect(validateSpy).toHaveBeenCalledWith({ + expect(validateModelsSpy).toHaveBeenCalledWith({ provider: "Alibaba Coding Plan", apiKey: "sk-custom-key", - baseUrl: "https://my-proxy.com/v1", - model: "qwen3.5-plus", + modelsUrl: "https://my-proxy.com/v1/models", signal: undefined, }); }); diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 1ed9d20fc..0517b7455 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -427,6 +427,100 @@ describe("Anthropic request fingerprint alignment", () => { expect(capturedBeta).toContain("mid-conversation-system-2026-04-07"); }); + it("adds the extended-cache-ttl beta to API-key requests that default to 1h caching", async () => { + const captureBeta = () => { + let captured: string | undefined; + const fetchMock = (async (_input: string | URL | Request, init?: RequestInit) => { + captured = (init?.headers as Record | undefined)?.["anthropic-beta"]; + return new Response( + JSON.stringify({ type: "error", error: { type: "invalid_request_error", message: "captured" } }), + { status: 400, headers: { "Content-Type": "application/json" } }, + ); + }) as typeof fetch; + return { fetchMock, beta: () => captured ?? "" }; + }; + const cacheContext: Context = { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], + }; + + const canonical = captureBeta(); + await streamAnthropic(ANTHROPIC_MODEL, cacheContext, { + apiKey: "sk-ant-api-test", + fetch: canonical.fetchMock, + }).result(); + expect(canonical.beta()).toContain("extended-cache-ttl-2025-04-11"); + + // Endpoints without long-cache support never send `ttl: "1h"`, so the + // companion beta must stay off the wire too. + const proxy = captureBeta(); + await streamAnthropic(UMANS_ANTHROPIC_MODEL, cacheContext, { + apiKey: "sk-umans-test", + fetch: proxy.fetchMock, + }).result(); + expect(proxy.beta()).not.toContain("extended-cache-ttl-2025-04-11"); + }); + + it("gates the effort beta and field off google-vertex requests (#5614)", async () => { + let capturedBeta: string | undefined; + let capturedBody: + | { + output_config?: { effort?: unknown }; + fallbacks?: Array<{ + model: string; + max_tokens?: number; + output_config?: { effort?: unknown }; + }>; + } + | undefined; + const fetchMock = (async (_input: string | URL | Request, init?: RequestInit) => { + capturedBeta = (init?.headers as Record | undefined)?.["anthropic-beta"]; + capturedBody = JSON.parse(String(init?.body ?? "{}")); + return new Response( + JSON.stringify({ type: "error", error: { type: "invalid_request_error", message: "captured" } }), + { status: 400, headers: { "Content-Type": "application/json" } }, + ); + }) as typeof fetch; + // Claude on Vertex uses api "anthropic-messages" and the rawPredict adapter, + // which rejects any `anthropic-beta` HTTP header value it doesn't understand. + // The effort beta must ride the body (`anthropic_beta`) instead — since this + // path can't deliver it there, primary and fallback effort fields are dropped. + const vertexModel: Model<"anthropic-messages"> = buildModel({ + ...ANTHROPIC_MODEL_SPEC, + id: "claude-haiku-4-5@20260101", + name: "Claude Haiku via Vertex", + provider: "google-vertex", + baseUrl: + "https://us-east5-aiplatform.googleapis.com/v1/projects/p/locations/us-east5/publishers/anthropic/models/claude-haiku-4-5:rawPredict", + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + }, + }); + + await streamAnthropic( + vertexModel, + { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }] }, + { + apiKey: "vertex-adc", + thinkingEnabled: true, + effort: "high", + fetch: fetchMock, + fallbacks: [ + { + model: "claude-sonnet-4-6@20260101", + max_tokens: 4_096, + output_config: { effort: "high" }, + }, + ], + }, + ).result(); + + expect(capturedBeta ?? "").not.toContain("effort-2025-11-24"); + expect(capturedBody?.output_config?.effort).toBeUndefined(); + expect(capturedBody?.fallbacks).toEqual([{ model: "claude-sonnet-4-6@20260101", max_tokens: 4_096 }]); + }); + it("adds the context-management beta to API-key thinking requests", async () => { let capturedBeta: string | undefined; const fetchMock = (async (_input: string | URL | Request, init?: RequestInit) => { @@ -504,7 +598,8 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.system).toEqual([ { type: "text", text: "stable system" }, - { type: "text", text: "stable durable context", cache_control: { type: "ephemeral" } }, + // Canonical Anthropic API-key requests default to the 1h breakpoint. + { type: "text", text: "stable durable context", cache_control: { type: "ephemeral", ttl: "1h" } }, ]); }); @@ -582,6 +677,48 @@ describe("Anthropic request fingerprint alignment", () => { expect(headers.Authorization).toBe("Bearer sk-ant-oat-test"); }); + it("honors opted-in OAuth fingerprint headers on non-official endpoints (#5888)", () => { + const options = buildAnthropicClientOptions({ + model: buildModel({ + ...ANTHROPIC_MODEL_SPEC, + provider: "custom-anthropic", + baseUrl: "https://proxy.example.com/anthropic", + headers: { + "anthropic-beta": "custom-beta-token", + "x-app": "custom-app-token", + "X-Stainless-Runtime-Version": "custom-runtime-token", + Authorization: "should-not-leak", + }, + compat: { allowAnthropicHeaderOverrides: true }, + }), + apiKey: "sk-ant-oat-test", + stream: true, + }); + + expect(options.defaultHeaders["anthropic-beta"]).toBe("custom-beta-token"); + expect(options.defaultHeaders["x-app"]).toBe("custom-app-token"); + expect(options.defaultHeaders["X-Stainless-Runtime-Version"]).toBe("custom-runtime-token"); + expect(options.defaultHeaders.Authorization).toBe("Bearer sk-ant-oat-test"); + }); + + it("keeps OAuth fingerprint defaults on official endpoints despite the compat opt-in", () => { + const headers = buildAnthropicHeaders({ + apiKey: "sk-ant-oat-test", + baseUrl: "https://api.anthropic.com", + isOAuth: true, + allowAnthropicHeaderOverrides: true, + modelHeaders: { + "anthropic-beta": "custom-beta-token", + "x-app": "custom-app-token", + "X-Stainless-Runtime-Version": "custom-runtime-token", + }, + }); + + expect(headers["anthropic-beta"]).not.toBe("custom-beta-token"); + expect(headers["x-app"]).toBe("cli"); + expect(headers["X-Stainless-Runtime-Version"]).toBe("v24.3.0"); + }); + it("suppresses the client-level X-Api-Key when model.headers carries a custom Authorization (#3391)", () => { const options = buildAnthropicClientOptions({ model: buildModel({ @@ -1682,13 +1819,14 @@ describe("Anthropic request fingerprint alignment", () => { FOUNDRY_BASE_URL: "https://foundry.example.com/anthropic/", ANTHROPIC_CUSTOM_HEADERS: "user-id: alice, x-route: engineering", }, - () => { + async () => { const options = buildAnthropicClientOptions({ model: ANTHROPIC_MODEL, apiKey: "foundry-token", extraBetas: [], stream: true, interleavedThinking: false, + hasTools: true, dynamicHeaders: {}, }); @@ -1697,6 +1835,24 @@ describe("Anthropic request fingerprint alignment", () => { expect(options.defaultHeaders["X-Api-Key"]).toBeUndefined(); expect(options.defaultHeaders["user-id"]).toBe("alice"); expect(options.defaultHeaders["x-route"]).toBe("engineering"); + expect(options.defaultHeaders["anthropic-beta"] ?? "").not.toContain( + "fine-grained-tool-streaming-2025-05-14", + ); + + const payload = await captureAnthropicPayload(ANTHROPIC_MODEL, { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: 0 }], + tools: [ + { + name: "ping", + description: "ping", + parameters: { type: "object", properties: {}, additionalProperties: false }, + }, + ], + }); + expect(payload).toMatchObject({ + tools: [expect.not.objectContaining({ eager_input_streaming: expect.anything() })], + }); }, ); }); diff --git a/packages/ai/test/anthropic-retry.test.ts b/packages/ai/test/anthropic-retry.test.ts index 03fe45c0d..9f20af608 100644 --- a/packages/ai/test/anthropic-retry.test.ts +++ b/packages/ai/test/anthropic-retry.test.ts @@ -83,6 +83,16 @@ describe("isProviderRetryableError", () => { ).toBe(false); expect(isProviderRetryableError(new Error("usage_limit_reached"))).toBe(false); expect(isProviderRetryableError(new Error("You have hit your ChatGPT usage limit"))).toBe(false); + // Anthropic monthly spend-cap 429 (issue #4787): must not retry, or the + // provider loop burns its budget on minutes-long retry-after backoff and + // surfaces "Deadline exceeded" instead of the quota error. + expect( + isProviderRetryableError( + new Error( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s monthly spend limit. Please try again later."}}', + ), + ), + ).toBe(false); // A generic transient rate limit (no account/usage framing) still retries. expect(isProviderRetryableError(new Error("Rate limit exceeded"))).toBe(true); }); diff --git a/packages/ai/test/anthropic-stream-envelope.test.ts b/packages/ai/test/anthropic-stream-envelope.test.ts index df97a902d..5701e60df 100644 --- a/packages/ai/test/anthropic-stream-envelope.test.ts +++ b/packages/ai/test/anthropic-stream-envelope.test.ts @@ -8,6 +8,7 @@ import { } from "@oh-my-pi/pi-ai/providers/anthropic-client"; import type { AssistantMessageEvent, Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { withEnv } from "./helpers"; const model: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4-5", @@ -1380,4 +1381,45 @@ describe("anthropic stream envelope handling", () => { expect(cacheControls[1]).toEqual({ type: "ephemeral" }); expect(cacheControls[2]).toEqual({ type: "ephemeral" }); }); + + it("defaults API-key requests to 1h cache TTL where long retention is supported", async () => { + type CapturedParams = { messages: Array<{ content: unknown }> }; + const payloads: CapturedParams[] = []; + vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation((params: unknown) => { + // Params captured verbatim at the mocked SDK boundary. + const captured = params as CapturedParams; + payloads.push(captured); + return createMockRequest(createTextSuccessEvents("ok")) as never; + }); + const proxyModel = buildModel({ + ...model, + compat: { ...model.compatConfig, supportsLongCacheRetention: false }, + } as ModelSpec<"anthropic-messages">); + const drain = async (testModel: Model<"anthropic-messages">): Promise => { + const stream = streamAnthropic(testModel, context, { apiKey: "sk-ant-test" }); + for await (const _ of stream) { + // drain stream + } + await stream.result(); + }; + + await drain(model); + await drain(proxyModel); + await withEnv({ PI_CACHE_RETENTION: "short" }, () => drain(model)); + + const cacheControls = payloads.map(payload => { + const content = payload.messages.at(-1)?.content; + if (!Array.isArray(content)) return undefined; + const lastBlock: { cache_control?: { ttl?: string; type: string } } | undefined = content.at(-1); + return lastBlock?.cache_control; + }); + // Agent sessions idle past 5 minutes on background jobs; the canonical + // Anthropic API defaults to the 1h breakpoint so resume doesn't cold-miss + // the whole prefix. + expect(cacheControls[0]).toEqual({ type: "ephemeral", ttl: "1h" }); + // Endpoints without long-cache support keep the plain 5m breakpoint. + expect(cacheControls[1]).toEqual({ type: "ephemeral" }); + // PI_CACHE_RETENTION=short opts back out of the 1h default. + expect(cacheControls[2]).toEqual({ type: "ephemeral" }); + }); }); diff --git a/packages/ai/test/auth-gateway-model-list.test.ts b/packages/ai/test/auth-gateway-model-list.test.ts new file mode 100644 index 000000000..59d0da81e --- /dev/null +++ b/packages/ai/test/auth-gateway-model-list.test.ts @@ -0,0 +1,38 @@ +import { expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { startAuthGateway } from "@oh-my-pi/pi-ai/auth-gateway"; +import { AuthStorage } from "@oh-my-pi/pi-ai/auth-storage"; +import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; + +test("model listing exposes one provider-qualified route per upstream model", async () => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), "gw-model-list-")); + const storage = await AuthStorage.create(path.join(dir, "auth.db")); + const anthropic = createMockModel({ provider: "anthropic", id: "shared-model" }); + const devin = createMockModel({ provider: "devin", id: "shared-model" }); + const handle = startAuthGateway({ + bind: "127.0.0.1:0", + bearerTokens: [], + storage, + resolveModel: () => undefined, + listModels: () => [anthropic, anthropic, devin, devin], + version: "test", + }); + + try { + const response = await fetch(`${handle.url}/v1/models`); + expect(response.status).toBe(200); + expect(await response.json()).toEqual({ + object: "list", + data: [ + { id: "anthropic/shared-model", object: "model", owned_by: "anthropic", api: "mock" }, + { id: "devin/shared-model", object: "model", owned_by: "devin", api: "mock" }, + ], + }); + } finally { + await handle.close(); + storage.close(); + await fs.rm(dir, { recursive: true, force: true }); + } +}); diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts index 84b0969db..d59cc3512 100644 --- a/packages/ai/test/auth-retry.test.ts +++ b/packages/ai/test/auth-retry.test.ts @@ -70,6 +70,7 @@ describe("isAuthRetryableError", () => { // credentials won't help an org/global limit. expect(isAuthRetryableError(Object.assign(new Error("429 too many requests"), { status: 429 }))).toBe(false); expect(isAuthRetryableError("Error: 401 unauthorized")).toBe(true); + expect(isAuthRetryableError("Encountered invalidated oauth token for user, failing request")).toBe(true); // xAI SuperGrok surfaces account exhaustion as 403 + "run out of credits" / // spending-limit, not 429. Must rotate so multi-account xai-oauth pools work. expect( @@ -312,6 +313,26 @@ describe("withAuth", () => { expect(keys).toEqual(["k0"]); }); + it("switches credentials when OpenRouter exhausts the daily free-model allowance", async () => { + const keys: string[] = []; + const result = await withAuth( + ctx => (ctx.error === undefined || !ctx.lastChance ? "exhausted-key" : "healthy-key"), + async key => { + keys.push(key); + if (key === "healthy-key") return "success"; + throw Object.assign( + new Error( + "429 Rate limit exceeded: free-models-per-day. Add 10 credits to unlock 1000 free model requests per day", + ), + { status: 429 }, + ); + }, + ); + + expect(result).toBe("success"); + expect(keys).toEqual(["exhausted-key", "healthy-key"]); + }); + it("stops retrying when the resolver returns undefined", async () => { const keys: string[] = []; const original = authError(); @@ -492,6 +513,25 @@ describe("withOAuthAccess", () => { ]); }); + it("invalidates and rotates directly when upstream reports an invalidated OAuth token", async () => { + const storage = fakeStorage({ + initial: access("dead", { credentialId: 1 }), + rotated: access("sibling", { credentialId: 2 }), + }); + const attempts: string[] = []; + const result = await withOAuthAccess(storage, "prov", async a => { + attempts.push(a.accessToken); + if (a.accessToken === "dead") { + throw new Error("Encountered invalidated oauth token for user, failing request"); + } + return "ok"; + }); + + expect(result).toBe("ok"); + expect(attempts).toEqual(["dead", "sibling"]); + expect(storage.calls).toEqual([{ forceRefresh: undefined }, "rotate", { forceRefresh: undefined }]); + }); + it("rotates directly to a sibling on usage limits", async () => { const storage = fakeStorage({ initial: access("dead"), diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 4c2dedd69..99238cda9 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -227,6 +227,48 @@ describe("AuthStorage codex oauth ranking", () => { expectExclusivePreference(counts, "api-acct-near", "api-acct-far"); }); + test("keeps a Codex session pinned after >1h idle", async () => { + if (!authStorage) throw new Error("test setup failed"); + const storage = authStorage; + + await storage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-pinned", "pinned@example.com") }, + { type: "oauth", ...createCredential("acct-sibling", "sibling@example.com") }, + ]); + + const base = Date.now(); + let clockOffset = 0; + vi.spyOn(Date, "now").mockImplementation(() => base + clockOffset); + + const setUsage = (pinnedPrimary: number, siblingPrimary: number): void => { + usageByAccount.set( + "acct-pinned", + createCodexUsageReport({ + accountId: "acct-pinned", + primary: { usedFraction: pinnedPrimary, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.5, resetInMs: 5 * 24 * HOUR_MS }, + }), + ); + usageByAccount.set( + "acct-sibling", + createCodexUsageReport({ + accountId: "acct-sibling", + primary: { usedFraction: siblingPrimary, resetInMs: HOUR_MS }, + secondary: { usedFraction: 0.5, resetInMs: 5 * 24 * HOUR_MS }, + }), + ); + }; + + setUsage(0.2, 0.9); + expect(await storage.getApiKey("openai-codex", "codex-idle-boundary")).toBe("api-acct-pinned"); + + // Codex long retention can preserve a prompt cache for 24h, so the + // Anthropic-specific 1h gate must not re-rank this still-usable pin. + setUsage(0.9, 0.2); + clockOffset = 2 * HOUR_MS; + expect(await storage.getApiKey("openai-codex", "codex-idle-boundary")).toBe("api-acct-pinned"); + }); + test("prefers fresh 5h ticker account at 0% usage", async () => { if (!authStorage) throw new Error("test setup failed"); @@ -1450,35 +1492,95 @@ describe("AuthStorage codex oauth ranking", () => { test.each([ ["gpt-5.6-terra", "free", "enterprise"], ["gpt-5.6-terra-pro", "go", "pro"], - ])("%s keeps a less-used %s account in ordinary ranking ahead of %s", async (modelId, lowUsagePlan, highUsagePlan) => { + ])( + "%s keeps a less-used %s account in ordinary ranking ahead of %s", + async (modelId, lowUsagePlan, highUsagePlan) => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-low-usage", "low-usage@example.com") }, + { type: "oauth", ...createCredential("acct-high-usage", "high-usage@example.com") }, + ]); + + usageByAccount.set( + "acct-low-usage", + createCodexUsageReport({ + accountId: "acct-low-usage", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: lowUsagePlan, email: "low-usage@example.com" }, + }), + ); + usageByAccount.set( + "acct-high-usage", + createCodexUsageReport({ + accountId: "acct-high-usage", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: highUsagePlan, email: "high-usage@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId }); + expect(apiKey).toBe("api-acct-low-usage"); + }, + ); + + test("keeps an eligible Codex session credential when usage headroom makes its sibling rank better", async () => { if (!authStorage) throw new Error("test setup failed"); - await authStorage.set("openai-codex", [ - { type: "oauth", ...createCredential("acct-low-usage", "low-usage@example.com") }, - { type: "oauth", ...createCredential("acct-high-usage", "high-usage@example.com") }, - ]); + const modelId = "gpt-5.6-sol"; + const sessionId = "codex-sticky-usage-rerank"; + const accounts = [ + { id: "acct-sticky-usage-a", email: "sticky-usage-a@example.com" }, + { id: "acct-sticky-usage-b", email: "sticky-usage-b@example.com" }, + ]; + const reportByAccount: Record = {}; + const setUsedFraction = (report: UsageReport, usedFraction: number): void => { + const used = usedFraction * 100; + for (const limit of report.limits) { + limit.amount.used = used; + limit.amount.remaining = 100 - used; + limit.amount.usedFraction = usedFraction; + limit.amount.remainingFraction = 1 - usedFraction; + limit.status = usedFraction >= 1 ? "exhausted" : usedFraction >= 0.9 ? "warning" : "ok"; + } + }; - usageByAccount.set( - "acct-low-usage", - createCodexUsageReport({ - accountId: "acct-low-usage", - primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, - secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, - metadata: { planType: lowUsagePlan, email: "low-usage@example.com" }, - }), - ); - usageByAccount.set( - "acct-high-usage", - createCodexUsageReport({ - accountId: "acct-high-usage", - primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, - secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, - metadata: { planType: highUsagePlan, email: "high-usage@example.com" }, - }), - ); + const base = Date.now(); + let clockOffset = 0; + vi.spyOn(Date, "now").mockImplementation(() => base + clockOffset); - const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId }); - expect(apiKey).toBe("api-acct-low-usage"); + await authStorage.set( + "openai-codex", + accounts.map(account => ({ type: "oauth", ...createCredential(account.id, account.email) })), + ); + for (const account of accounts) { + const report = createCodexUsageReport({ + accountId: account.id, + primary: { usedFraction: 0.25, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.25, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "business", email: account.email }, + }); + reportByAccount[account.id] = report; + usageByAccount.set(account.id, report); + } + + const firstApiKey = await authStorage.getApiKey("openai-codex", sessionId, { modelId }); + if (!firstApiKey) throw new Error("expected initial Codex credential"); + const stickyAccount = firstApiKey.replace(/^api-/, ""); + const siblingAccount = stickyAccount === accounts[0]!.id ? accounts[1]!.id : accounts[0]!.id; + const stickyReport = reportByAccount[stickyAccount]; + const siblingReport = reportByAccount[siblingAccount]; + if (!stickyReport || !siblingReport) throw new Error("expected reports for both Codex accounts"); + + setUsedFraction(stickyReport, 0.85); + setUsedFraction(siblingReport, 0.01); + // Step past the usage-report TTL so the second resolve re-fetches the + // inverted headroom instead of ranking on the cached first-resolve reports + // (mirrors mid-session header ingest / TTL expiry in a real session). + clockOffset = 10 * 60 * 1000; + expect(await authStorage.getApiKey("openai-codex", sessionId, { modelId })).toBe(firstApiKey); }); test("reranks a Terra session on a Go account when it switches to Sol", async () => { @@ -2020,7 +2122,7 @@ function createClaudeLimit(args: { key: "5h" | "7d"; durationMs: number; usedFraction: number; - resetInMs: number; + resetInMs?: number; tier?: "fable"; }): UsageLimit { const clamped = Math.min(Math.max(args.usedFraction, 0), 1); @@ -2038,7 +2140,7 @@ function createClaudeLimit(args: { id: args.key, label, durationMs: args.durationMs, - resetsAt: Date.now() + args.resetInMs, + ...(args.resetInMs === undefined ? {} : { resetsAt: Date.now() + args.resetInMs }), }, amount: { unit: "percent", @@ -2054,9 +2156,9 @@ function createClaudeLimit(args: { function createClaudeUsageReport(args: { accountId: string; - primary: { usedFraction: number; resetInMs: number }; - secondary: { usedFraction: number; resetInMs: number }; - fableSecondary?: { usedFraction: number; resetInMs: number }; + primary: { usedFraction: number; resetInMs?: number }; + secondary: { usedFraction: number; resetInMs?: number }; + fableSecondary?: { usedFraction: number; resetInMs?: number }; }): UsageReport { const limits = [ createClaudeLimit({ @@ -2163,6 +2265,35 @@ describe("AuthStorage claude oauth ranking", () => { expectExclusivePreference(counts, "api-acct-near", "api-acct-far"); }); + test("assumes the full duration remains when ranking clockless windows", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-clockless", "clockless@example.com") }, + { type: "oauth", ...createCredential("acct-clocked", "clocked@example.com") }, + ]); + + usageByAccount.set( + "acct-clockless", + createClaudeUsageReport({ + accountId: "acct-clockless", + primary: { usedFraction: 0 }, + secondary: { usedFraction: 0 }, + }), + ); + usageByAccount.set( + "acct-clocked", + createClaudeUsageReport({ + accountId: "acct-clocked", + primary: { usedFraction: 0, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.05, resetInMs: 22 * HOUR_MS }, + }), + ); + + const apiKey = await authStorage.getApiKey("anthropic", "session-claude-clockless"); + expect(apiKey).toBe("api-acct-clocked"); + }); + test("resolves equal-priority accounts to one deterministic pick", async () => { if (!authStorage) throw new Error("test setup failed"); @@ -2396,4 +2527,88 @@ describe("AuthStorage claude oauth ranking", () => { const apiKey = await authStorage.getApiKey("anthropic", "session-claude-single"); expect(apiKey).toBe("api-acct-solo"); }); + + test("re-ranks a session pinned to a now-worse account after >1h of Anthropic idle", async () => { + if (!authStorage) throw new Error("test setup failed"); + const storage = authStorage; + + await storage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-pinned", "pinned@example.com") }, + { type: "oauth", ...createCredential("acct-fresh", "fresh@example.com") }, + ]); + + const base = Date.now(); + let clockOffset = 0; + vi.spyOn(Date, "now").mockImplementation(() => base + clockOffset); + + // t0: acct-pinned is healthy; acct-fresh's 5h window is hot (>=85%), + // so ranking picks acct-pinned and pins the session to it. + const setUsage = (pinnedPrimary: number, freshPrimary: number): void => { + usageByAccount.set( + "acct-pinned", + createClaudeUsageReport({ + accountId: "acct-pinned", + primary: { usedFraction: pinnedPrimary, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.5, resetInMs: 5 * 24 * HOUR_MS }, + }), + ); + usageByAccount.set( + "acct-fresh", + createClaudeUsageReport({ + accountId: "acct-fresh", + primary: { usedFraction: freshPrimary, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.5, resetInMs: 5 * 24 * HOUR_MS }, + }), + ); + }; + + setUsage(0.2, 0.9); + expect(await storage.getApiKey("anthropic", "claude-idle-gating")).toBe("api-acct-pinned"); + + // The tables turn: acct-pinned's 5h window is now hot, acct-fresh is cool. + setUsage(0.9, 0.2); + + // Within 1h of the last resolve the conversation prefix is plausibly warm, + // so the pin must hold even though it is now the worse account. + clockOffset = 30 * 60 * 1000; + expect(await storage.getApiKey("anthropic", "claude-idle-gating")).toBe("api-acct-pinned"); + + // After >1h of Anthropic request inactivity the prompt cache is no longer + // guaranteed warm, so ranking must run again and rotate to the better sibling. + clockOffset = 30 * 60 * 1000 + 2 * HOUR_MS; + expect(await storage.getApiKey("anthropic", "claude-idle-gating")).toBe("api-acct-fresh"); + }); + + test("keeps the pinned account after idle when siblings rank equal (tie-break)", async () => { + if (!authStorage) throw new Error("test setup failed"); + const storage = authStorage; + + await storage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-a", "a@example.com") }, + { type: "oauth", ...createCredential("acct-b", "b@example.com") }, + ]); + + const base = Date.now(); + let clockOffset = 0; + vi.spyOn(Date, "now").mockImplementation(() => base + clockOffset); + + for (const accountId of ["acct-a", "acct-b"]) { + usageByAccount.set( + accountId, + createClaudeUsageReport({ + accountId, + primary: { usedFraction: 0.25, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.25, resetInMs: 4 * 24 * HOUR_MS }, + }), + ); + } + + const first = await storage.getApiKey("anthropic", "claude-idle-tie"); + expect(first).toBeDefined(); + + // Past the warm window ranking runs again, but both accounts score equal, + // so the pin must win the tie rather than churn to the sibling. + clockOffset = HOUR_MS + 1; + expect(await storage.getApiKey("anthropic", "claude-idle-tie")).toBe(first); + }); }); diff --git a/packages/ai/test/auth-storage-codex-workspace-identity.test.ts b/packages/ai/test/auth-storage-codex-workspace-identity.test.ts new file mode 100644 index 000000000..1f4b38e73 --- /dev/null +++ b/packages/ai/test/auth-storage-codex-workspace-identity.test.ts @@ -0,0 +1,305 @@ +/** + * OpenAI Codex workspace-scoped credential identity. + * + * The ChatGPT workspace (`chatgpt_account_id`, captured as `orgId` at login) + * is the subscription pool a Codex token draws limits from. One email can + * hold a personal Plus/Pro plan plus Team/Enterprise seats — different + * workspaces with independent pools — while every member of one workspace + * shares the workspace id. Identity therefore composes `email|org:`: + * - same email + same workspace => replace in place (re-login); + * - same email + diff workspace => coexist (personal + enterprise seat); + * - diff email + same workspace => coexist (two Team members, #197); + * - workspace-less legacy rows keep their bare email key and are claimed + * in place by the first workspace-scoped login with the same email. + */ +import { Database } from "bun:sqlite"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { + type AuthCredential, + type AuthCredentialStore, + AuthStorage, + SqliteAuthCredentialStore, + type StoredAuthCredential, +} from "@oh-my-pi/pi-ai/auth-storage"; +import type { UsageReport } from "@oh-my-pi/pi-ai/usage"; +import * as codexUsage from "@oh-my-pi/pi-ai/usage/openai-codex"; +import { removeWithRetries } from "../../utils/src/temp"; + +const EMAIL = "shared@example.com"; +const PERSONAL_WS = "ws-personal-1111"; +const TEAM_WS = "ws-team-2222"; + +function codexCredential(args: { + suffix: string; + accountId: string; + /** Workspace qualifier; omitted for legacy rows written before workspace capture. */ + orgId?: string; + orgName?: string; + email?: string; +}): AuthCredential { + return { + type: "oauth", + access: `access-${args.suffix}`, + refresh: `refresh-${args.suffix}`, + expires: Date.now() + 3_600_000, + accountId: args.accountId, + email: args.email ?? EMAIL, + orgId: args.orgId, + orgName: args.orgName, + }; +} + +function readIdentityRows(dbPath: string): Array<{ identity_key: string | null; disabled_cause: string | null }> { + const db = new Database(dbPath, { readonly: true }); + try { + return db + .prepare( + "SELECT identity_key, disabled_cause FROM auth_credentials WHERE provider = 'openai-codex' ORDER BY id ASC", + ) + .all() as Array<{ identity_key: string | null; disabled_cause: string | null }>; + } finally { + db.close(); + } +} + +describe("openai-codex workspace-scoped credential identity", () => { + let tempDir = ""; + let dbPath = ""; + let store: SqliteAuthCredentialStore | null = null; + + beforeEach(async () => { + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-codex-ws-identity-")); + dbPath = path.join(tempDir, "agent.db"); + store = await SqliteAuthCredentialStore.open(dbPath); + }); + + afterEach(async () => { + store?.close(); + store = null; + if (tempDir) await removeWithRetries(tempDir); + }); + + it("stores a personal plan and an enterprise seat of one email side by side and updates same-workspace logins in place", () => { + if (!store) throw new Error("test setup failed"); + + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "personal", accountId: PERSONAL_WS, orgId: PERSONAL_WS, orgName: "plus" }), + ); + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "team", accountId: TEAM_WS, orgId: TEAM_WS, orgName: "enterprise" }), + ); + + expect(readIdentityRows(dbPath)).toEqual([ + { identity_key: `email:${EMAIL}|org:${PERSONAL_WS}`, disabled_cause: null }, + { identity_key: `email:${EMAIL}|org:${TEAM_WS}`, disabled_cause: null }, + ]); + + // Same-workspace re-login: replaces the matching row instead of adding a third. + const rows = store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "team-renewed", accountId: TEAM_WS, orgId: TEAM_WS, orgName: "enterprise" }), + ); + expect(readIdentityRows(dbPath)).toEqual([ + { identity_key: `email:${EMAIL}|org:${PERSONAL_WS}`, disabled_cause: null }, + { identity_key: `email:${EMAIL}|org:${TEAM_WS}`, disabled_cause: null }, + ]); + const teamRow = rows.find(row => row.credential.type === "oauth" && row.credential.orgId === TEAM_WS); + expect(teamRow?.credential.type).toBe("oauth"); + if (teamRow?.credential.type === "oauth") { + expect(teamRow.credential.access).toBe("access-team-renewed"); + } + }); + + it("keeps two members of one workspace separate even though they share the workspace id", () => { + if (!store) throw new Error("test setup failed"); + + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "alice", accountId: TEAM_WS, orgId: TEAM_WS, email: "alice@example.com" }), + ); + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "bob", accountId: TEAM_WS, orgId: TEAM_WS, email: "bob@example.com" }), + ); + + expect(readIdentityRows(dbPath)).toEqual([ + { identity_key: `email:alice@example.com|org:${TEAM_WS}`, disabled_cause: null }, + { identity_key: `email:bob@example.com|org:${TEAM_WS}`, disabled_cause: null }, + ]); + }); + + it("upgrades a legacy email-keyed row on the first workspace-scoped login with the same email", () => { + if (!store) throw new Error("test setup failed"); + + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "legacy", accountId: PERSONAL_WS }), + ); + expect(readIdentityRows(dbPath)).toEqual([{ identity_key: `email:${EMAIL}`, disabled_cause: null }]); + + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "team", accountId: TEAM_WS, orgId: TEAM_WS, orgName: "team" }), + ); + expect(readIdentityRows(dbPath)).toEqual([ + { identity_key: `email:${EMAIL}|org:${TEAM_WS}`, disabled_cause: null }, + ]); + }); + + it("never clobbers workspace-scoped rows with a workspace-less credential", () => { + if (!store) throw new Error("test setup failed"); + + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "personal", accountId: PERSONAL_WS, orgId: PERSONAL_WS, orgName: "plus" }), + ); + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "team", accountId: TEAM_WS, orgId: TEAM_WS, orgName: "enterprise" }), + ); + store.upsertAuthCredentialForProvider( + "openai-codex", + codexCredential({ suffix: "orgless", accountId: PERSONAL_WS }), + ); + + expect(readIdentityRows(dbPath)).toEqual([ + { identity_key: `email:${EMAIL}|org:${PERSONAL_WS}`, disabled_cause: null }, + { identity_key: `email:${EMAIL}|org:${TEAM_WS}`, disabled_cause: null }, + { identity_key: `email:${EMAIL}`, disabled_cause: null }, + ]); + }); +}); + +// ─── Usage report dedupe partitioning ─────────────────────────────────────── + +interface CacheEntry { + value: string; + expiresAtSec: number; +} + +function makeStore(rows: StoredAuthCredential[]): AuthCredentialStore { + const cache = new Map(); + return { + close() {}, + listAuthCredentials() { + return rows; + }, + updateAuthCredential() {}, + deleteAuthCredential() {}, + tryDisableAuthCredentialIfMatches() { + return false; + }, + replaceAuthCredentialsForProvider() { + return rows; + }, + upsertAuthCredentialForProvider() { + return rows; + }, + deleteAuthCredentialsForProvider() {}, + getCache(key) { + const entry = cache.get(key); + if (!entry) return null; + if (entry.expiresAtSec * 1000 <= Date.now()) return null; + return entry.value; + }, + setCache(key, value, expiresAtSec) { + cache.set(key, { value, expiresAtSec }); + }, + cleanExpiredCache() {}, + }; +} + +function codexRow( + id: number, + args?: { orgId?: string; orgName?: string; accountId?: string; email?: string }, +): StoredAuthCredential { + return { + id, + provider: "openai-codex", + credential: { + type: "oauth", + access: `oat-${id}`, + refresh: `refresh-${id}`, + expires: Date.now() + 3_600_000, + accountId: args?.accountId ?? args?.orgId ?? "ws-legacy", + email: args?.email ?? EMAIL, + orgId: args?.orgId, + orgName: args?.orgName, + }, + disabledCause: null, + }; +} + +/** Report carrying ONLY email identity — workspace attribution must come from the credential. */ +function emailOnlyReport(email: string): UsageReport { + return { + provider: "openai-codex", + fetchedAt: Date.now(), + limits: [ + { + id: "openai-codex:primary", + label: "5 hours", + scope: { provider: "openai-codex", windowId: "5h" }, + window: { id: "5h", label: "5 hours" }, + amount: { used: 42, limit: 100, unit: "percent" }, + status: "ok", + }, + ], + metadata: { email }, + }; +} + +describe("openai-codex usage report dedupe partitions by workspace", () => { + let storage: AuthStorage | null = null; + + afterEach(() => { + storage?.close(); + storage = null; + vi.restoreAllMocks(); + }); + + it("keeps reports from two workspaces on one email separate and attributes each to its workspace", async () => { + storage = new AuthStorage( + makeStore([ + codexRow(1, { orgId: PERSONAL_WS, orgName: "plus" }), + codexRow(2, { orgId: TEAM_WS, orgName: "enterprise" }), + ]), + { + usageProviderResolver: provider => + provider === "openai-codex" ? codexUsage.openaiCodexUsageProvider : undefined, + }, + ); + await storage.reload(); + + vi.spyOn(codexUsage.openaiCodexUsageProvider, "fetchUsage").mockImplementation(async () => + emailOnlyReport(EMAIL), + ); + + const reports = ((await storage.fetchUsageReports()) ?? []).filter(r => r.provider === "openai-codex"); + expect(reports).toHaveLength(2); + const orgIds = reports.map(report => report.metadata?.orgId).sort(); + expect(orgIds).toEqual([PERSONAL_WS, TEAM_WS].sort()); + const orgNames = reports.map(report => report.metadata?.orgName).sort(); + expect(orgNames).toEqual(["enterprise", "plus"].sort()); + }); + + it("still merges workspace-less reports with the same email into one row", async () => { + storage = new AuthStorage(makeStore([codexRow(1), codexRow(2)]), { + usageProviderResolver: provider => + provider === "openai-codex" ? codexUsage.openaiCodexUsageProvider : undefined, + }); + await storage.reload(); + + vi.spyOn(codexUsage.openaiCodexUsageProvider, "fetchUsage").mockImplementation(async () => + emailOnlyReport(EMAIL), + ); + + const reports = ((await storage.fetchUsageReports()) ?? []).filter(r => r.provider === "openai-codex"); + expect(reports).toHaveLength(1); + }); +}); diff --git a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts index 312a073dd..1a2e9c32f 100644 --- a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts +++ b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts @@ -285,6 +285,146 @@ describe("AuthStorage OAuth refresh race", () => { expect(refreshCalls).toBe(1); }); + test("serializes rotating provider refresh tokens across AuthStorage instances", async () => { + if (!authStorage) throw new Error("test setup failed"); + + const expires = Date.now() - 60_000; + const refreshedExpires = Date.now() + 60 * 60_000; + const usedRefreshTokens = new Set(); + let refreshCalls = 0; + + oauthUtils.registerOAuthProvider({ + id: "unit-oauth-cross-process", + name: "Unit OAuth Cross Process", + sourceId: "auth-storage-oauth-refresh-race-test", + async login() { + return { access: "unused", refresh: "unused", expires: refreshedExpires }; + }, + async refreshToken(credentials) { + refreshCalls += 1; + if (usedRefreshTokens.has(credentials.refresh)) { + throw new Error('HTTP 400 invalid_grant {"error":"invalid_grant"}'); + } + usedRefreshTokens.add(credentials.refresh); + await Bun.sleep(50); + return { + ...credentials, + access: "access-rotated", + refresh: "refresh-rotated", + expires: refreshedExpires, + }; + }, + getApiKey(credentials) { + return credentials.access; + }, + }); + + await authStorage.set("unit-oauth-cross-process", [ + { type: "oauth", access: "access-old", refresh: "refresh-old", expires }, + ]); + + const secondStore = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db")); + const secondStorage = new AuthStorage(secondStore); + await secondStorage.reload(); + try { + const [first, second] = await Promise.all([ + authStorage.getApiKey("unit-oauth-cross-process", "session-first"), + secondStorage.getApiKey("unit-oauth-cross-process", "session-second"), + ]); + + expect(first).toBe("access-rotated"); + expect(second).toBe("access-rotated"); + expect(refreshCalls).toBe(1); + expect(secondStore.listAuthCredentials("unit-oauth-cross-process")).toHaveLength(1); + } finally { + secondStorage.close(); + } + }); + + test("does not overwrite a peer rotation after releasing the refresh lease", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + const sharedStore = store; + + const expires = Date.now() - 60_000; + const refreshedExpires = Date.now() + 60 * 60_000; + let credentialId: number | undefined; + + oauthUtils.registerOAuthProvider({ + id: "unit-oauth-post-lease-race", + name: "Unit OAuth Post-Lease Race", + sourceId: "auth-storage-oauth-refresh-race-test", + async login() { + return { access: "unused", refresh: "unused", expires: refreshedExpires }; + }, + async refreshToken(credentials) { + return { + ...credentials, + access: "access-from-this-process", + refresh: "refresh-from-this-process", + expires: refreshedExpires, + }; + }, + getApiKey(credentials) { + if (credentialId === undefined) throw new Error("credential id not initialized"); + sharedStore.updateAuthCredential(credentialId, { + type: "oauth", + access: "access-from-peer", + refresh: "refresh-from-peer", + expires: refreshedExpires, + }); + return credentials.access; + }, + }); + + await authStorage.set("unit-oauth-post-lease-race", [ + { type: "oauth", access: "access-old", refresh: "refresh-old", expires }, + ]); + credentialId = store.listAuthCredentials("unit-oauth-post-lease-race")[0]?.id; + expect(credentialId).toBeDefined(); + + const apiKey = await authStorage.getApiKey("unit-oauth-post-lease-race", "session-post-lease"); + expect(apiKey).toBe("access-from-this-process"); + const persisted = store.listAuthCredentials("unit-oauth-post-lease-race")[0]?.credential; + expect(persisted?.type).toBe("oauth"); + if (persisted?.type === "oauth") { + expect(persisted.refresh).toBe("refresh-from-peer"); + expect(persisted.access).toBe("access-from-peer"); + } + }); + + test("returns the targeted OAuth row after a compare-and-set refresh loss", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + + const expires = Date.now() - 60_000; + await authStorage.set("unit-oauth-cas-loss", [ + { type: "oauth", access: "access-first", refresh: "refresh-first", expires }, + { type: "oauth", access: "access-target", refresh: "refresh-target", expires }, + ]); + const credentialId = store.listAuthCredentials("unit-oauth-cas-loss")[1]?.id; + expect(credentialId).toBeDefined(); + if (credentialId === undefined) return; + + const result = await authStorage.refreshStoredOAuthCredential("unit-oauth-cas-loss", { + credentialId, + forceRefresh: true, + credentialFromRow: credential => credential, + async refresh(current) { + store!.updateAuthCredential(credentialId, { + ...current, + access: "access-from-peer", + refresh: "refresh-from-peer", + }); + return { ...current, access: "access-from-this-process", refresh: "refresh-from-this-process" }; + }, + }); + + expect(result.refreshed).toBe(false); + expect(result.credential).toMatchObject({ access: "access-from-peer", refresh: "refresh-from-peer" }); + const rows = store.listAuthCredentials("unit-oauth-cas-loss"); + expect(rows[0]?.credential).toMatchObject({ type: "oauth", access: "access-first" }); + expect(rows[1]?.credential).toMatchObject({ type: "oauth", access: "access-from-peer" }); + }); + test("syncs peer-updated SQLite OAuth rows before returning access tokens", async () => { if (!authStorage || !store) throw new Error("test setup failed"); diff --git a/packages/ai/test/auth-storage-usage-cache.test.ts b/packages/ai/test/auth-storage-usage-cache.test.ts index e6dda593d..fad014b51 100644 --- a/packages/ai/test/auth-storage-usage-cache.test.ts +++ b/packages/ai/test/auth-storage-usage-cache.test.ts @@ -531,39 +531,23 @@ describe("AuthStorage usage cache: header ingestion", () => { }); describe("AuthStorage usage cache: terminal refresh failure", () => { - // Regression: a revoked refresh token used to fail the in-line OAuth refresh - // inside the usage probe, get silently swallowed, then trigger the upstream - // 401 → null → last-good fallback chain. The credential was therefore never - // removed from the candidate set and the /usage TUI kept rendering yesterday's - // report — including its now-elapsed `resetsAt`, which the renderer printed - // as e.g. `(-612090ms)`. The fix CAS-disables the row on a definitive refresh - // failure and clears the cache, so the credential drops out cleanly. - it("disables credential and suppresses last-good when OAuth refresh fails with invalid_grant", async () => { - // Row whose access token has just expired — within the 60s refresh skew so - // the usage probe is forced to refresh before issuing the upstream call. + // Usage polling is non-critical: refresh failure must not disable a + // credential whose current access token can still satisfy the probe. + it("keeps credential and probes with current access after a definitive refresh failure", async () => { const row = oauthRow(1, "a@example.com"); - (row.credential as { expires: number }).expires = Date.now() - 1000; + if (row.credential.type !== "oauth") throw new Error("expected OAuth test credential"); + row.credential.expires = Date.now() + 30_000; const rows = [row]; - - // `makeStore` returns `false` from `tryDisableAuthCredentialIfMatches`, - // which would short-circuit our disable. Use a local store that actually - // performs the soft-delete so we can observe the AuthStorage-side effects. const cache = new Map(); let disableCalls = 0; const store: ObservableStore = { cache, close() {}, - listAuthCredentials: () => rows.filter(r => !r.disabledCause), + listAuthCredentials: () => rows.filter(candidate => !candidate.disabledCause), updateAuthCredential() {}, - deleteAuthCredential(id: number, cause: string) { - const target = rows.find(r => r.id === id); - if (target) target.disabledCause = cause; - }, - tryDisableAuthCredentialIfMatches(id: number, _data: string, cause: string) { + deleteAuthCredential() {}, + tryDisableAuthCredentialIfMatches() { disableCalls += 1; - const target = rows.find(r => r.id === id); - if (!target) return false; - target.disabledCause = cause; return true; }, replaceAuthCredentialsForProvider: () => rows, @@ -581,16 +565,6 @@ describe("AuthStorage usage cache: terminal refresh failure", () => { cleanExpiredCache() {}, }; - // Pre-populate the cache with a "last good" report whose inner expiresAt - // is in the past (so `get()` misses) but the entry is still reachable via - // `getStale()`. Mirrors what the prior poll would have written. - const lastGood = makeReport("a@example.com"); - const cacheKey = "usage_cache:report:2:anthropic:default:oauth|account:account-1|email:a@example.com"; - cache.set(cacheKey, { - value: JSON.stringify({ value: lastGood, expiresAt: 1 }), - expiresAtSec: Math.floor((Date.now() + 24 * 60 * 60_000) / 1000), - }); - const storage = new AuthStorage(store, { usageProviderResolver: provider => (provider === "anthropic" ? claudeUsage.claudeUsageProvider : undefined), refreshOAuthCredential: async () => { @@ -599,31 +573,48 @@ describe("AuthStorage usage cache: terminal refresh failure", () => { }); await storage.reload(); - const fetchSpy = vi.spyOn(claudeUsage.claudeUsageProvider, "fetchUsage"); - + const fetchSpy = vi + .spyOn(claudeUsage.claudeUsageProvider, "fetchUsage") + .mockResolvedValue(makeReport("a@example.com")); try { const reports = anthropicReports(await storage.fetchUsageReports()); - // No last-good fallback: the row was disabled before lastGood could leak. - expect(reports).toHaveLength(0); - // CAS disable was attempted exactly once on the failing row. - expect(disableCalls).toBe(1); - expect(rows[0].disabledCause).toContain("invalid_grant"); - // Upstream probe is short-circuited — no point asking the provider - // with a credential we've just torn down. - expect(fetchSpy).not.toHaveBeenCalled(); - // Cache entry was neutralized: a future `getStale` lookup (e.g. on - // re-login under the same account identity) returns null, not the - // stale report with its already-elapsed `resetsAt`. - const rawAfter = cache.get(cacheKey); - expect(rawAfter).toBeDefined(); - const parsedAfter = JSON.parse(rawAfter!.value); - expect(parsedAfter.value).toBeNull(); - // And a second poll surfaces nothing — the credential is gone from - // `listAuthCredentials`, so `#collectUsageRequests` doesn't even - // look it up. - const secondPoll = anthropicReports(await storage.fetchUsageReports()); - expect(secondPoll).toHaveLength(0); + expect(reports).toHaveLength(1); + expect(reports[0]?.metadata?.email).toBe("a@example.com"); + expect(disableCalls).toBe(0); + expect(rows[0]?.disabledCause).toBeNull(); + expect(fetchSpy).toHaveBeenCalledTimes(1); + expect(fetchSpy.mock.calls[0]?.[0].credential.accessToken).toBe("oat-1"); + } finally { + storage.close(); + vi.restoreAllMocks(); + } + }); + + it("suppresses last-good fallback when an expired OAuth access token has a definitive refresh failure", async () => { + const row = oauthRow(3, "expired@example.com"); + if (row.credential.type !== "oauth") throw new Error("expected OAuth test credential"); + row.credential.expires = Date.now() - 1000; + const store = makeStore([row]); + const cacheKey = "usage_cache:report:2:anthropic:default:oauth|account:account-3|email:expired@example.com"; + store.cache.set(cacheKey, { + value: JSON.stringify({ value: makeReport("expired@example.com"), expiresAt: 1 }), + expiresAtSec: Math.floor((Date.now() + 24 * 60 * 60_000) / 1000), + }); + const storage = new AuthStorage(store, { + usageProviderResolver: provider => (provider === "anthropic" ? claudeUsage.claudeUsageProvider : undefined), + refreshOAuthCredential: async () => { + throw new Error("OAuth refresh failed: 400 invalid_grant: refresh token revoked"); + }, + }); + await storage.reload(); + const fetchSpy = vi.spyOn(claudeUsage.claudeUsageProvider, "fetchUsage").mockResolvedValue(null); + try { + expect(anthropicReports(await storage.fetchUsageReports())).toHaveLength(0); + expect(fetchSpy).toHaveBeenCalledTimes(1); + expect(row.disabledCause).toBeNull(); + const cached = JSON.parse(store.cache.get(cacheKey)!.value); + expect(cached.value).toBeNull(); } finally { storage.close(); vi.restoreAllMocks(); diff --git a/packages/ai/test/auth-storage-xai-oauth-usage.test.ts b/packages/ai/test/auth-storage-xai-oauth-usage.test.ts new file mode 100644 index 000000000..d20317c77 --- /dev/null +++ b/packages/ai/test/auth-storage-xai-oauth-usage.test.ts @@ -0,0 +1,144 @@ +import { describe, expect, it } from "bun:test"; +import { type AuthCredentialStore, AuthStorage, type StoredAuthCredential } from "@oh-my-pi/pi-ai/auth-storage"; +import type { UsageFetchParams, UsageProvider } from "@oh-my-pi/pi-ai/usage"; +import { withEnv } from "./helpers"; + +function makeStore(credentials: StoredAuthCredential[] = []): AuthCredentialStore { + const cache = new Map(); + return { + close() {}, + listAuthCredentials() { + return credentials; + }, + updateAuthCredential() {}, + deleteAuthCredential() {}, + tryDisableAuthCredentialIfMatches() { + return false; + }, + replaceAuthCredentialsForProvider() { + return []; + }, + upsertAuthCredentialForProvider() { + return []; + }, + deleteAuthCredentialsForProvider() {}, + getCache(key) { + const entry = cache.get(key); + if (!entry || entry.expiresAtSec * 1000 <= Date.now()) return null; + return entry.value; + }, + setCache(key, value, expiresAtSec) { + cache.set(key, { value, expiresAtSec }); + }, + cleanExpiredCache() {}, + }; +} + +function captureUsageProvider(calls: UsageFetchParams[]): UsageProvider { + return { + id: "xai-oauth", + supports: params => params.credential.type === "oauth" && !!params.credential.accessToken, + async fetchUsage(params) { + calls.push(params); + return { + provider: "xai-oauth", + fetchedAt: Date.now(), + limits: [], + }; + }, + }; +} + +describe("xAI OAuth environment usage", () => { + it("treats dedicated XAI_OAUTH_TOKEN as an OAuth bearer for SuperGrok usage", async () => { + const calls: UsageFetchParams[] = []; + await withEnv({ XAI_OAUTH_TOKEN: "oauth-bearer", XAI_API_KEY: undefined }, async () => { + const storage = new AuthStorage(makeStore(), { + usageProviderResolver: provider => (provider === "xai-oauth" ? captureUsageProvider(calls) : undefined), + }); + await storage.reload(); + + await storage.fetchUsageReports(); + }); + + expect(calls).toHaveLength(1); + expect(calls[0]?.credential).toEqual({ type: "oauth", accessToken: "oauth-bearer" }); + }); + + it("prefers stored OAuth credentials to XAI_OAUTH_TOKEN", async () => { + const calls: UsageFetchParams[] = []; + await withEnv({ XAI_OAUTH_TOKEN: "env-oauth-bearer" }, async () => { + const storage = new AuthStorage( + makeStore([ + { + id: 1, + provider: "xai-oauth", + credential: { + type: "oauth", + access: "stored-oauth-bearer", + refresh: "stored-refresh-token", + expires: Date.now() + 3_600_000, + }, + disabledCause: null, + }, + ]), + { + usageProviderResolver: provider => (provider === "xai-oauth" ? captureUsageProvider(calls) : undefined), + }, + ); + await storage.reload(); + + await storage.fetchUsageReports(); + }); + + expect(calls).toHaveLength(1); + expect(calls[0]?.credential).toEqual({ + type: "oauth", + accessToken: "stored-oauth-bearer", + refreshToken: "stored-refresh-token", + expiresAt: expect.any(Number), + }); + }); + + it("uses XAI_OAUTH_TOKEN when stored xAI credentials contain only an API key", async () => { + const calls: UsageFetchParams[] = []; + await withEnv({ XAI_OAUTH_TOKEN: "env-oauth-bearer" }, async () => { + const storage = new AuthStorage( + makeStore([ + { + id: 1, + provider: "xai-oauth", + credential: { + type: "api_key", + key: "stored-api-key", + }, + disabledCause: null, + }, + ]), + { + usageProviderResolver: provider => (provider === "xai-oauth" ? captureUsageProvider(calls) : undefined), + }, + ); + await storage.reload(); + + await storage.fetchUsageReports(); + }); + + expect(calls).toHaveLength(1); + expect(calls[0]?.credential).toEqual({ type: "oauth", accessToken: "env-oauth-bearer" }); + }); + + it("does not send shared XAI_API_KEY to the SuperGrok usage endpoint", async () => { + const calls: UsageFetchParams[] = []; + await withEnv({ XAI_OAUTH_TOKEN: undefined, XAI_API_KEY: "paid-api-key" }, async () => { + const storage = new AuthStorage(makeStore(), { + usageProviderResolver: provider => (provider === "xai-oauth" ? captureUsageProvider(calls) : undefined), + }); + await storage.reload(); + + await storage.fetchUsageReports(); + }); + + expect(calls).toEqual([]); + }); +}); diff --git a/packages/ai/test/claude-usage-retry.test.ts b/packages/ai/test/claude-usage-retry.test.ts index 831e9a8f7..490e85970 100644 --- a/packages/ai/test/claude-usage-retry.test.ts +++ b/packages/ai/test/claude-usage-retry.test.ts @@ -24,7 +24,7 @@ function baseParams() { credential: { type: "oauth" as const, accessToken: "oat-test", - accountId: "org_test", + accountId: "account_test", email: "user@example.com", expiresAt: Date.now() + 60_000, }, @@ -68,6 +68,16 @@ describe("claudeUsageProvider retry contract", () => { expect(attempt).toBe(2); }); + it("does not treat the organization response header as account identity", async () => { + const fetchMock = (async () => + jsonResponse(200, VALID_PAYLOAD, { "anthropic-organization-id": "org_header" })) as FetchImpl; + + const report = await claudeUsageProvider.fetchUsage(baseParams(), makeContext(fetchMock)); + + expect(report?.metadata?.accountId).toBe("account_test"); + expect(report?.metadata?.orgId).toBeUndefined(); + }); + it("does NOT retry on 401 — permanent for this credential", async () => { let attempt = 0; const fetchMock = (async () => { diff --git a/packages/ai/test/cursor-exec-handlers.test.ts b/packages/ai/test/cursor-exec-handlers.test.ts index e2b2d268e..354f6be05 100644 --- a/packages/ai/test/cursor-exec-handlers.test.ts +++ b/packages/ai/test/cursor-exec-handlers.test.ts @@ -18,6 +18,7 @@ import { type AgentRunRequest, AgentServerMessageSchema, ExecServerMessageSchema, + McpArgsSchema, ReadArgsSchema, } from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; @@ -434,6 +435,7 @@ function newBlockState(): BlockState { get currentToolCall() { return toolCall; }, + resolvedMcpToolCallIds: new Set(), firstTokenTime: undefined, setTextBlock: b => { textBlock = b; @@ -514,6 +516,60 @@ describe("Cursor exec local-work tracking (issue #4593)", () => { expect(written.length).toBe(1); }); + it("marks an MCP call as resolved before its streamed block arrives", async () => { + const output = cursorAssistantMessage(); + const stream = new AssistantMessageEventStream(); + const state = newBlockState(); + const h2Request = { write: () => true } as unknown as Parameters[5]; + const serverMsg = create(AgentServerMessageSchema, { + message: { + case: "execServerMessage", + value: create(ExecServerMessageSchema, { + id: 1, + execId: "exec-mcp-1", + message: { + case: "mcpArgs", + value: create(McpArgsSchema, { + name: "mcp__fixture_report", + toolName: "mcp__fixture_report", + toolCallId: "call-mcp-1", + providerIdentifier: "pi-agent", + }), + }, + }), + }, + }); + const execHandlers: CursorExecHandlers = { + async mcp(args) { + return { + role: "toolResult", + toolCallId: args.toolCallId, + toolName: args.toolName, + content: [{ type: "text", text: "reported" }], + isError: false, + timestamp: 1, + }; + }, + }; + + await handleServerMessage( + serverMsg, + output, + stream, + state, + new Map(), + h2Request, + execHandlers, + undefined, + { + sawTokenDelta: false, + }, + [], + ); + + expect(state.resolvedMcpToolCallIds.has("call-mcp-1")).toBe(true); + }); + it("survives a local exec tool outliving the lazy idle budget end to end", async () => { const workDone = Promise.withResolvers(); // The tracked work completes only once the lazy watchdog has consulted diff --git a/packages/ai/test/cursor-h2-transport-error.test.ts b/packages/ai/test/cursor-h2-transport-error.test.ts new file mode 100644 index 000000000..81d3c799d --- /dev/null +++ b/packages/ai/test/cursor-h2-transport-error.test.ts @@ -0,0 +1,37 @@ +import { describe, expect, it } from "bun:test"; +import { ProviderResponseError } from "@oh-my-pi/pi-ai/error"; +import { mapH2TransportError } from "@oh-my-pi/pi-ai/providers/cursor"; + +const BASE_URL = "https://api2.cursor.sh"; + +describe("mapH2TransportError", () => { + it("rewrites the opaque bun ALPN failure into an actionable Cursor error", () => { + const raw = Object.assign(new Error("h2 is not supported"), { code: "ERR_HTTP2_ERROR" }); + const mapped = mapH2TransportError(raw, BASE_URL); + expect(mapped).toBeInstanceOf(ProviderResponseError); + const err = mapped as ProviderResponseError; + expect(err.provider).toBe("cursor"); + expect(err.kind).toBe("runtime"); + expect(err.message).toContain(BASE_URL); + expect(err.message).toContain("ALPN"); + expect(err.message).toContain("providers.cursor.baseUrl"); + expect(err.cause).toBe(raw); + }); + + it("matches the h2-not-supported message case-insensitively", () => { + const raw = Object.assign(new Error("H2 Is Not Supported"), { code: "ERR_HTTP2_ERROR" }); + expect(mapH2TransportError(raw, BASE_URL)).toBeInstanceOf(ProviderResponseError); + }); + + it("passes through an HTTP/2 error whose message is unrelated to ALPN", () => { + const raw = Object.assign(new Error("Stream closed with error code NGHTTP2_INTERNAL_ERROR"), { + code: "ERR_HTTP2_ERROR", + }); + expect(mapH2TransportError(raw, BASE_URL)).toBe(raw); + }); + + it("passes through a non-HTTP/2 error even when it mentions h2", () => { + const raw = Object.assign(new Error("h2 is not supported"), { code: "ECONNRESET" }); + expect(mapH2TransportError(raw, BASE_URL)).toBe(raw); + }); +}); diff --git a/packages/ai/test/cursor-streaming-args.test.ts b/packages/ai/test/cursor-streaming-args.test.ts index df81beadc..8447914e1 100644 --- a/packages/ai/test/cursor-streaming-args.test.ts +++ b/packages/ai/test/cursor-streaming-args.test.ts @@ -8,7 +8,7 @@ import { type UsageState, } from "@oh-my-pi/pi-ai/providers/cursor"; import type { AssistantMessage, AssistantMessageEvent } from "@oh-my-pi/pi-ai/types"; -import { getStreamingPartialJson } from "@oh-my-pi/pi-ai/utils/block-symbols"; +import { getStreamingPartialJson, kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; interface Harness { @@ -58,6 +58,7 @@ function newHarness(): Harness { get currentToolCall() { return toolCall; }, + resolvedMcpToolCallIds: new Set(), firstTokenTime: undefined, setTextBlock: b => { textBlock = b; @@ -172,6 +173,19 @@ describe("mergeCursorMcpToolCallArgs", () => { }); }); +describe("Cursor MCP exec resolution", () => { + it("marks a streamed MCP call already resolved by the exec bridge", () => { + const h = newHarness(); + h.state.resolvedMcpToolCallIds.add("call-resolved"); + + startMcpToolCall(h, "mcp__fixture_report", "call-resolved"); + + const block = h.output.content[0] as ToolCallState; + expect(block[kCursorExecResolved]).toBe(true); + expect(h.state.resolvedMcpToolCallIds.size).toBe(0); + }); +}); + describe("processInteractionUpdate content block ordering", () => { it("opens a new text block after a completed tool call", () => { const h = newHarness(); diff --git a/packages/ai/test/cursor-terminal-error.test.ts b/packages/ai/test/cursor-terminal-error.test.ts new file mode 100644 index 000000000..b25abdec6 --- /dev/null +++ b/packages/ai/test/cursor-terminal-error.test.ts @@ -0,0 +1,257 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as http2 from "node:http2"; +import { create, toBinary } from "@bufbuild/protobuf"; +import { streamCursor } from "@oh-my-pi/pi-ai/providers/cursor"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { + AgentServerMessageSchema, + InteractionUpdateSchema, + TextDeltaUpdateSchema, + TurnEndedUpdateSchema, +} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; + +const CONNECT_END_STREAM_FLAG = 0b00000010; + +type Scenario = + | { kind: "success" } + | { kind: "connect-error-after-turn" } + | { kind: "grpc-trailer-after-turn" } + | { kind: "end-before-turn" } + | { kind: "hang-after-turn" }; + +let server: http2.Http2Server | undefined; +const sessions = new Set(); +let scenario: Scenario = { kind: "success" }; + +function frameConnectMessage(data: Uint8Array, flags = 0): Buffer { + const frame = Buffer.alloc(5 + data.length); + frame[0] = flags; + frame.writeUInt32BE(data.length, 1); + frame.set(data, 5); + return frame; +} + +function textDeltaFrame(text: string): Buffer { + const message = create(AgentServerMessageSchema, { + message: { + case: "interactionUpdate", + value: create(InteractionUpdateSchema, { + message: { + case: "textDelta", + value: create(TextDeltaUpdateSchema, { text }), + }, + }), + }, + }); + return frameConnectMessage(toBinary(AgentServerMessageSchema, message)); +} + +function turnEndedFrame(): Buffer { + const message = create(AgentServerMessageSchema, { + message: { + case: "interactionUpdate", + value: create(InteractionUpdateSchema, { + message: { + case: "turnEnded", + value: create(TurnEndedUpdateSchema, {}), + }, + }), + }, + }); + return frameConnectMessage(toBinary(AgentServerMessageSchema, message)); +} + +function connectEndErrorFrame(code: string, message: string): Buffer { + const payload = Buffer.from(JSON.stringify({ error: { code, message } }), "utf8"); + return frameConnectMessage(payload, CONNECT_END_STREAM_FLAG); +} + +async function startServer(): Promise { + server = http2.createServer(); + server.on("session", session => { + sessions.add(session); + session.on("close", () => sessions.delete(session)); + }); + server.on("stream", (stream: http2.ServerHttp2Stream, headers: http2.IncomingHttpHeaders) => { + stream.on("data", () => {}); + + if (headers[":path"] !== "/agent.v1.AgentService/Run") { + stream.respond({ ":status": 404 }); + stream.end(); + return; + } + + if (scenario.kind === "grpc-trailer-after-turn") { + stream.respond( + { + ":status": 200, + "content-type": "application/connect+proto", + }, + { waitForTrailers: true }, + ); + stream.on("wantTrailers", () => { + stream.sendTrailers({ + "grpc-status": "13", + "grpc-message": encodeURIComponent("post-turn trailer failure"), + }); + }); + stream.write(textDeltaFrame("hello")); + stream.write(turnEndedFrame()); + stream.end(); + return; + } + + stream.respond({ + ":status": 200, + "content-type": "application/connect+proto", + }); + + if (scenario.kind === "end-before-turn") { + stream.write(textDeltaFrame("partial")); + stream.end(); + return; + } + + stream.write(Buffer.concat([textDeltaFrame("hello"), turnEndedFrame()])); + + if (scenario.kind === "connect-error-after-turn") { + stream.write(connectEndErrorFrame("unavailable", "post-turn connect failure")); + stream.end(); + return; + } + + if (scenario.kind === "hang-after-turn") { + return; + } + + stream.end(); + }); + + const listening = Promise.withResolvers(); + server.once("error", listening.reject); + server.listen(0, "127.0.0.1", listening.resolve); + await listening.promise; + const address = server.address(); + if (!address || typeof address === "string") { + throw new Error("expected http2 fixture server to bind a tcp port"); + } + return `http://127.0.0.1:${address.port}`; +} + +function makeModel(baseUrl: string): Model<"cursor-agent"> { + return buildModel({ + id: "cursor-terminal-fixture", + name: "Cursor terminal fixture", + api: "cursor-agent", + provider: "cursor", + baseUrl, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1, + maxTokens: 1, + }); +} + +const context: Context = { + messages: [{ role: "user", content: "terminal lifecycle", timestamp: 1 }], +}; + +async function collectStream(model: Model<"cursor-agent">, options?: { signal?: AbortSignal }) { + const stream = streamCursor(model, context, { apiKey: "test-token", signal: options?.signal }); + const eventTypes: string[] = []; + for await (const event of stream) { + eventTypes.push(event.type); + } + const result = await stream.result(); + return { eventTypes, result }; +} + +async function stopServer(): Promise { + for (const session of sessions) { + session.destroy(); + } + sessions.clear(); + if (!server) return; + const closing = server; + server = undefined; + const closed = Promise.withResolvers(); + closing.close(error => { + if (error) { + closed.reject(error); + } else { + closed.resolve(); + } + }); + await closed.promise; +} + +afterEach(async () => { + scenario = { kind: "success" }; + await stopServer(); +}); + +describe("Cursor terminal lifecycle after turnEnded", () => { + it("emits done only after turnEnded and a clean protocol end", async () => { + scenario = { kind: "success" }; + const baseUrl = await startServer(); + const { eventTypes, result } = await collectStream(makeModel(baseUrl)); + expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); + expect(result.stopReason).toBe("stop"); + expect(result.errorMessage).toBeUndefined(); + }); + + it("surfaces CONNECT end-stream errors that arrive after turnEnded", async () => { + scenario = { kind: "connect-error-after-turn" }; + const baseUrl = await startServer(); + const { eventTypes, result } = await collectStream(makeModel(baseUrl)); + expect(eventTypes[0]).toBe("start"); + expect(eventTypes.at(-1)).toBe("error"); + expect(eventTypes).not.toContain("done"); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("Connect error unavailable: post-turn connect failure"); + }); + + it("surfaces nonzero gRPC trailers that arrive after turnEnded", async () => { + scenario = { kind: "grpc-trailer-after-turn" }; + const baseUrl = await startServer(); + const { eventTypes, result } = await collectStream(makeModel(baseUrl)); + expect(eventTypes[0]).toBe("start"); + expect(eventTypes.at(-1)).toBe("error"); + expect(eventTypes).not.toContain("done"); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("gRPC error 13: post-turn trailer failure"); + }); + + it("rejects when the stream ends before turnEnded", async () => { + scenario = { kind: "end-before-turn" }; + const baseUrl = await startServer(); + const { eventTypes, result } = await collectStream(makeModel(baseUrl)); + expect(eventTypes[0]).toBe("start"); + expect(eventTypes.at(-1)).toBe("error"); + expect(eventTypes).not.toContain("done"); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("Cursor stream ended before turnEnded"); + }); + + it("aborts without emitting done when the signal fires", async () => { + scenario = { kind: "hang-after-turn" }; + const baseUrl = await startServer(); + const controller = new AbortController(); + const stream = streamCursor(makeModel(baseUrl), context, { + apiKey: "test-token", + signal: controller.signal, + }); + const eventTypes: string[] = []; + for await (const event of stream) { + eventTypes.push(event.type); + if (event.type === "text_delta") controller.abort(); + } + const result = await stream.result(); + expect(eventTypes[0]).toBe("start"); + expect(eventTypes.at(-1)).toBe("error"); + expect(eventTypes).not.toContain("done"); + expect(result.stopReason).toBe("aborted"); + }); +}); diff --git a/packages/ai/test/devin-history.test.ts b/packages/ai/test/devin-history.test.ts new file mode 100644 index 000000000..f6c1f2e13 --- /dev/null +++ b/packages/ai/test/devin-history.test.ts @@ -0,0 +1,111 @@ +import { describe, expect, it } from "bun:test"; +import { gunzipSync } from "node:zlib"; +import { create, fromBinary, toBinary } from "@bufbuild/protobuf"; +import { streamDevin } from "@oh-my-pi/pi-ai/providers/devin"; +import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { GetChatMessageRequestSchema } from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/api_server_pb/api_server_pb"; +import { GetUserJwtResponseSchema } from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/auth_pb/auth_pb"; + +const devinModel: Model<"devin-agent"> = buildModel({ + id: "devin-test", + name: "Devin Test", + api: "devin-agent", + provider: "devin", + baseUrl: "https://server.codeium.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 64_000, +}); + +const zeroUsage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function assistant(overrides: Partial): AssistantMessage { + return { + role: "assistant", + content: [], + api: "devin-agent", + provider: "devin", + model: "devin-test", + usage: zeroUsage, + stopReason: "stop", + timestamp: 1, + ...overrides, + }; +} + +async function captureRequest(context: Context) { + const authPayload = toBinary(GetUserJwtResponseSchema, create(GetUserJwtResponseSchema, { userJwt: "jwt" })); + let requestPayload: Uint8Array | undefined; + const fetchImpl = (async (input: string | URL | Request, init?: RequestInit) => { + if (String(input).includes("GetUserJwt")) return new Response(authPayload); + requestPayload = new Uint8Array(init?.body as ArrayBuffer); + return new Response(new Uint8Array()); + }) as typeof fetch; + + await streamDevin(devinModel, context, { apiKey: "token", fetch: fetchImpl }).result(); + if (!requestPayload) throw new Error("Devin chat request was not captured"); + const length = new DataView(requestPayload.buffer, requestPayload.byteOffset, requestPayload.byteLength).getUint32( + 1, + false, + ); + const compressed = requestPayload.subarray(5, 5 + length); + return fromBinary(GetChatMessageRequestSchema, gunzipSync(compressed)); +} + +describe("streamDevin history handoff", () => { + it("removes foreign provider metadata and empty aborted turns", async () => { + const context: Context = { + messages: [ + { role: "user", content: "start", timestamp: 1 }, + assistant({ + api: "openai-responses", + provider: "openai-codex", + model: "gpt-5.6-sol", + responseId: "resp_foreign", + content: [ + { type: "thinking", thinking: "foreign reasoning", thinkingSignature: '{"type":"reasoning"}' }, + { type: "text", text: "foreign answer" }, + ], + }), + { role: "user", content: "continue", timestamp: 2 }, + assistant({ + responseId: "bot-12345678-1234-4234-8234-123456789abc", + content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "native-signature" }], + }), + { role: "user", content: "interrupt", timestamp: 3 }, + assistant({ + api: "openai-responses", + provider: "openai-codex", + model: "gpt-5.6-sol", + stopReason: "aborted", + }), + { role: "user", content: "resume", timestamp: 4 }, + ], + }; + + const request = await captureRequest(context); + const foreign = request.chatMessagePrompts[1]; + const native = request.chatMessagePrompts[3]; + + expect(request.chatMessagePrompts).toHaveLength(6); + expect(foreign?.messageId).not.toBe("resp_foreign"); + expect(foreign?.messageId).toMatch(/^bot-[0-9a-f-]{36}$/); + expect(foreign?.prompt).toContain("foreign reasoning"); + expect(foreign?.prompt).toContain("foreign answer"); + expect(foreign?.thinking).toBe(""); + expect(foreign?.signature).toBe(""); + expect(native?.messageId).toBe("bot-12345678-1234-4234-8234-123456789abc"); + expect(native?.thinking).toBe("native reasoning"); + expect(native?.signature).toBe("native-signature"); + }); +}); diff --git a/packages/ai/test/devin-large-request-recovery.test.ts b/packages/ai/test/devin-large-request-recovery.test.ts new file mode 100644 index 000000000..da869df8c --- /dev/null +++ b/packages/ai/test/devin-large-request-recovery.test.ts @@ -0,0 +1,289 @@ +import { describe, expect, it } from "bun:test"; +import { create, toBinary } from "@bufbuild/protobuf"; +import * as AIError from "@oh-my-pi/pi-ai/error"; +import { streamDevin } from "@oh-my-pi/pi-ai/providers/devin"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { GetChatMessageResponseSchema } from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/api_server_pb/api_server_pb"; +import { GetUserJwtResponseSchema } from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/auth_pb/auth_pb"; + +const CONNECT_END_STREAM_FLAG = 0x02; +const LARGE_TOOL_RESULT_BYTES = 160 * 1024; + +const devinModel: Model<"devin-agent"> = buildModel({ + id: "devin-test", + name: "Devin Test", + api: "devin-agent", + provider: "devin", + baseUrl: "https://server.codeium.com", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 64_000, +}); + +const largeReadContext: Context = { + messages: Array.from({ length: 4 }, (_, index) => ({ + role: "toolResult" as const, + toolCallId: `read-${index}`, + toolName: "read", + content: [{ type: "text" as const, text: String(index).repeat(LARGE_TOOL_RESULT_BYTES) }], + isError: false, + timestamp: index + 1, + })), +}; +const largeFixedContext: Context = { + systemPrompt: ["s".repeat(320 * 1024)], + messages: [{ role: "user", content: "small history", timestamp: 1 }], + tools: [ + { + name: "large_tool", + description: "d".repeat(320 * 1024), + parameters: { type: "object", properties: {} }, + }, + ], +}; + +function connectFrame(payload: Uint8Array, flag = 0): Uint8Array { + const frame = Buffer.alloc(5 + payload.length); + frame[0] = flag; + frame.writeUInt32BE(payload.length, 1); + frame.set(payload, 5); + return frame; +} + +function connectErrorFrame(code: string, message: string): Uint8Array { + const payload = Buffer.from(JSON.stringify({ error: { code, message } }), "utf8"); + return connectFrame(payload, CONNECT_END_STREAM_FLAG); +} + +async function runTrailerError(context: Context, code: string, message: string, leadingFrames: Uint8Array[] = []) { + const authPayload = toBinary(GetUserJwtResponseSchema, create(GetUserJwtResponseSchema, { userJwt: "jwt" })); + const trailer = connectErrorFrame(code, message); + const fetchImpl = (async (input: string | URL | Request) => { + if (String(input).includes("GetUserJwt")) return new Response(authPayload); + return new Response( + new ReadableStream({ + start(controller) { + for (const frame of leadingFrames) controller.enqueue(frame); + controller.enqueue(trailer); + controller.close(); + }, + }), + { status: 200 }, + ); + }) as typeof fetch; + + return streamDevin(devinModel, context, { apiKey: "token", fetch: fetchImpl }).result(); +} + +describe("streamDevin large request recovery", () => { + it("classifies an opaque invalid_argument on cumulative large read output as context overflow", async () => { + const result = await runTrailerError( + largeReadContext, + "invalid_argument", + "an internal error occurred (trace ID: large-request)", + ); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("trace ID: large-request"); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(true); + }); + + it("keeps the same opaque trailer transient for a small request", async () => { + const result = await runTrailerError( + { messages: [{ role: "user", content: "hi", timestamp: 1 }] }, + "invalid_argument", + "an internal error occurred (trace ID: small-request)", + ); + + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(false); + expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(true); + }); + + it("does not compact a large system prompt and tool schema with small history", async () => { + const result = await runTrailerError( + largeFixedContext, + "invalid_argument", + "an internal error occurred (trace ID: fixed-request)", + ); + + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(false); + expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(true); + }); + + it("does not compact a large history after partial output", async () => { + const partial = toBinary( + GetChatMessageResponseSchema, + create(GetChatMessageResponseSchema, { deltaText: "partial" }), + ); + const result = await runTrailerError( + largeReadContext, + "invalid_argument", + "an internal error occurred (trace ID: partial-output)", + [connectFrame(partial)], + ); + + expect(result.content).toContainEqual({ type: "text", text: "partial" }); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(false); + expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(true); + }); + + it("does not reinterpret a specific invalid argument on a large request", async () => { + const result = await runTrailerError(largeReadContext, "invalid_argument", "unknown chat model uid"); + + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(false); + }); + + it("keeps the trailer transient for a huge active turn user message with no prior history", async () => { + const result = await runTrailerError( + { + messages: [ + { + role: "user" as const, + content: "u".repeat(520 * 1024), + timestamp: 1, + }, + ], + }, + "invalid_argument", + "an internal error occurred (trace ID: huge-user-request)", + ); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("trace ID: huge-user-request"); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(false); + expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(true); + }); + + it("classifies as context overflow if there is huge eligible prior history before the active turn user message", async () => { + const result = await runTrailerError( + { + messages: [ + { + role: "toolResult" as const, + toolCallId: "read-prior", + toolName: "read", + content: [{ type: "text" as const, text: "p".repeat(520 * 1024) }], + isError: false, + timestamp: 1, + }, + { + role: "user" as const, + content: "small user prompt", + timestamp: 2, + }, + ], + }, + "invalid_argument", + "an internal error occurred (trace ID: prior-history-overflow)", + ); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("trace ID: prior-history-overflow"); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(true); + }); + + it("classifies as context overflow if the request follows a large tool execution", async () => { + const result = await runTrailerError( + { + messages: [ + { + role: "user" as const, + content: "small user prompt", + timestamp: 1, + }, + { + role: "assistant" as const, + content: [ + { type: "text" as const, text: "running a tool" }, + { + type: "toolCall" as const, + id: "call-1", + name: "large_tool", + arguments: {}, + }, + ], + api: "devin-agent" as const, + provider: "devin" as const, + model: "devin-test", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse" as const, + timestamp: 2, + }, + { + role: "toolResult" as const, + toolCallId: "call-1", + toolName: "large_tool", + content: [{ type: "text" as const, text: "t".repeat(520 * 1024) }], + isError: false, + timestamp: 3, + }, + ], + }, + "invalid_argument", + "an internal error occurred (trace ID: tool-execution-overflow)", + ); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("trace ID: tool-execution-overflow"); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(true); + }); + it("keeps prior user-role execution history eligible before the active prompt", async () => { + const result = await runTrailerError( + { + messages: [ + { + role: "user" as const, + content: "execution output: ".concat("x".repeat(520 * 1024)), + timestamp: 1, + }, + { + role: "user" as const, + content: "small active prompt", + timestamp: 2, + }, + ], + }, + "invalid_argument", + "an internal error occurred (trace ID: user-role-history)", + ); + + expect(result.stopReason).toBe("error"); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(true); + }); + + it("keeps the trailer transient for a large current prompt split across multiple trailing user/developer messages", async () => { + const result = await runTrailerError( + { + messages: [ + { + role: "user" as const, + content: "u".repeat(520 * 1024), + timestamp: 1, + }, + { + role: "developer" as const, + content: "small notice/companion", + timestamp: 2, + }, + ], + }, + "invalid_argument", + "an internal error occurred (trace ID: split-active-prompt)", + ); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("trace ID: split-active-prompt"); + expect(AIError.is(result.errorId, AIError.Flag.ContextOverflow)).toBe(false); + expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(true); + }); +}); diff --git a/packages/ai/test/devin-usage.test.ts b/packages/ai/test/devin-usage.test.ts new file mode 100644 index 000000000..4b65b6e28 --- /dev/null +++ b/packages/ai/test/devin-usage.test.ts @@ -0,0 +1,66 @@ +import { describe, expect, it } from "bun:test"; +import { create, toBinary } from "@bufbuild/protobuf"; +import { streamDevin } from "@oh-my-pi/pi-ai/providers/devin"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { GetChatMessageResponseSchema } from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/api_server_pb/api_server_pb"; +import { GetUserJwtResponseSchema } from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/auth_pb/auth_pb"; +import { + ModelUsageStatsSchema, + StopReason, +} from "@oh-my-pi/pi-catalog/discovery/devin-gen/exa/codeium_common_pb/codeium_common_pb"; + +function frameConnectMessage(payload: Uint8Array): Uint8Array { + const out = new Uint8Array(5 + payload.length); + const view = new DataView(out.buffer); + view.setUint8(0, 0); + view.setUint32(1, payload.length, false); + out.set(payload, 5); + return out; +} + +const devinModel: Model<"devin-agent"> = buildModel({ + id: "devin-test", + name: "Devin Test", + api: "devin-agent", + provider: "devin", + baseUrl: "https://server.codeium.com", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1, + maxTokens: 1, +}); + +const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 1 }] }; + +describe("streamDevin usage", () => { + it("includes cached tokens in totalTokens", async () => { + const authPayload = toBinary(GetUserJwtResponseSchema, create(GetUserJwtResponseSchema, { userJwt: "jwt" })); + const response = create(GetChatMessageResponseSchema, { + messageId: "msg-1", + stopReason: StopReason.STOP_PATTERN, + usage: create(ModelUsageStatsSchema, { + inputTokens: 11n, + outputTokens: 7n, + cacheReadTokens: 100n, + cacheWriteTokens: 13n, + }), + }); + const responseFrame = frameConnectMessage(toBinary(GetChatMessageResponseSchema, response)); + const fetchImpl = (async (input: string | URL | Request) => { + if (String(input).includes("GetUserJwt")) return new Response(authPayload); + return new Response(responseFrame); + }) as typeof fetch; + + const result = await streamDevin(devinModel, context, { apiKey: "token", fetch: fetchImpl }).result(); + + expect(result.usage).toMatchObject({ + input: 11, + output: 7, + cacheRead: 100, + cacheWrite: 13, + totalTokens: 131, + }); + }); +}); diff --git a/packages/ai/test/error-aierr.test.ts b/packages/ai/test/error-aierr.test.ts index a9458255c..b2a4a918c 100644 --- a/packages/ai/test/error-aierr.test.ts +++ b/packages/ai/test/error-aierr.test.ts @@ -135,4 +135,18 @@ describe("aierr flag helpers", () => { expect(AIError.retriable(AIError.create(AIError.Flag.Transient))).toBe(true); expect(AIError.retriable(AIError.create(AIError.Flag.UsageLimit))).toBe(true); }); + + it("recognizes explicit transient stream parse diagnostics without broad text false positives", () => { + const message = "JSON Parse error: Unterminated string"; + expect(AIError.isTransientStreamParseError(new Error(message))).toBe(true); + expect(AIError.isTransientStreamParseError(new Error("response truncated"))).toBe(true); + expect(AIError.isTransientStreamParseError(message)).toBe(true); + expect(AIError.isTransientStreamParseError("Unexpected end of JSON input")).toBe(true); + expect(AIError.isTransientStreamParseError("JSON.parse: unexpected end of data at line 1 column 8")).toBe(true); + expect(AIError.isTransientStreamParseError("Unexpected EOF")).toBe(true); + expect(AIError.isTransientStreamParseError("EOF while parsing")).toBe(true); + expect(AIError.isTransientStreamParseError("truncated")).toBe(false); + expect(AIError.isTransientStreamParseError("end of file")).toBe(false); + expect(AIError.isTransientStreamParseError("Unexpected token in JSON")).toBe(false); + }); }); diff --git a/packages/ai/test/error-id.test.ts b/packages/ai/test/error-id.test.ts index 0b63fb970..7de72b343 100644 --- a/packages/ai/test/error-id.test.ts +++ b/packages/ai/test/error-id.test.ts @@ -43,6 +43,25 @@ describe("error-id classification", () => { expect(AIError.retriable(id)).toBe(true); }); + it("classifies provider connection failures as transient", () => { + const assistant = message({ + errorMessage: "Unable to connect. Is the computer able to access the url?", + }); + const id = AIError.classifyMessage(assistant); + expect(AIError.is(id, AIError.Flag.Transient)).toBe(true); + expect(AIError.retriable(id)).toBe(true); + }); + + it("keeps authenticated connection rejections non-retryable", () => { + const assistant = message({ + errorMessage: "Unable to connect: 401 Unauthorized", + }); + const id = AIError.classifyMessage(assistant); + expect(AIError.is(id, AIError.Flag.AuthFailed)).toBe(true); + expect(AIError.is(id, AIError.Flag.Transient)).toBe(false); + expect(AIError.retriable(id)).toBe(false); + }); + it("keeps provider content filters non-retryable", () => { const error = new AIError.ProviderResponseError("Provider returned error finish_reason: content_filter", { provider: "openrouter", diff --git a/packages/ai/test/github-copilot-openai-base-url.test.ts b/packages/ai/test/github-copilot-openai-base-url.test.ts index 0b65672df..76cffdaae 100644 --- a/packages/ai/test/github-copilot-openai-base-url.test.ts +++ b/packages/ai/test/github-copilot-openai-base-url.test.ts @@ -30,6 +30,13 @@ function getRequestHeader( return new Headers(init?.headers).get(headerName); } +async function getRequestBody(input: string | URL | Request, init?: RequestInit): Promise> { + if (input instanceof Request) { + return (await input.clone().json()) as Record; + } + return JSON.parse(String(init?.body)) as Record; +} + function createUnauthorizedResponse(): Response { return new Response(JSON.stringify({ error: { message: "Unauthorized" } }), { status: 401, @@ -79,6 +86,29 @@ describe("GitHub Copilot OpenAI transport base URL", () => { expect(requestedUrls[0]).toBe("https://api.githubcopilot.com/responses"); }); + it("omits OpenAI priority service tier while native OpenAI keeps it", async () => { + const requestedBodies: Record[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + requestedBodies.push(await getRequestBody(input, init)); + return createUnauthorizedResponse(); + }); + const requestOptions = { + apiKey: testToken, + fetch: fetchMock as unknown as typeof fetch, + serviceTier: "priority" as const, + }; + + const copilotModel = getBundledModel("github-copilot", "gpt-5.4") as Model<"openai-responses">; + await streamOpenAIResponses(copilotModel, testContext, requestOptions).result(); + + const openAIModel = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">; + await streamOpenAIResponses(openAIModel, testContext, requestOptions).result(); + + expect(requestedBodies).toHaveLength(2); + expect(requestedBodies[0]?.service_tier).toBeUndefined(); + expect(requestedBodies[1]?.service_tier).toBe("priority"); + }); + it("routes structured enterprise credentials to the enterprise chat completions host", async () => { const requestedUrls: string[] = []; const requestedAuthHeaders: Array = []; diff --git a/packages/ai/test/gitlab-duo-workflow-provider.test.ts b/packages/ai/test/gitlab-duo-workflow-provider.test.ts index 456231f8c..52a65c1f6 100644 --- a/packages/ai/test/gitlab-duo-workflow-provider.test.ts +++ b/packages/ai/test/gitlab-duo-workflow-provider.test.ts @@ -291,7 +291,6 @@ describe("GitLab Duo Workflow provider protocol", () => { it("builds startRequest goal as a bare ChatML transcript with tool-run linkage", () => { const patToken = `${"glpat"}-abcdefgh12345678ijkl`; const sessionCookie = "_gitlab_session=0123456789abcdef0123456789abcdef"; - const credentialTokens = [patToken, sessionCookie]; const replayContext: Context = { systemPrompt: [`OMP system instructions: preserve the local tool bridge. token ${patToken}`], @@ -378,10 +377,11 @@ describe("GitLab Duo Workflow provider protocol", () => { expect(payload.goal).not.toContain("call-1"); expect(payload.goal).not.toContain('"id":'); expect(payload.goal).not.toContain(" id="); - // Content is forwarded verbatim — the provider performs no credential redaction. - for (const token of credentialTokens) { - expect(payload.goal).toContain(token); - } + // Outbound credential redaction (#5655) scrubs plausible live credentials + // from the rendered transcript; low-entropy look-alikes pass through. + expect(payload.goal).not.toContain(patToken); + expect(payload.goal).toContain("[gitlab_token_redacted]"); + expect(payload.goal).toContain(sessionCookie); expect(payload.goal).not.toContain("[REDACTED]"); // Bare transcript: user content is emitted verbatim (no escaping, no boundary // declaration — that was the agreed "完全裸转录" design). A ChatML-breakout @@ -394,7 +394,8 @@ describe("GitLab Duo Workflow provider protocol", () => { // The OMP system prompt lives in the flow config system slot, not the goal. const flowPrompt = payload.flowConfig?.prompts[0]; expect(flowPrompt?.prompt_template.system).toContain("OMP system instructions: preserve the local tool bridge."); - expect(flowPrompt?.prompt_template.system).toContain(patToken); + expect(flowPrompt?.prompt_template.system).not.toContain(patToken); + expect(flowPrompt?.prompt_template.system).toContain("[gitlab_token_redacted]"); // This goal IS a multi-turn ChatML transcript, so the system slot appends the // history-note telling the model the `<|im_start|>`/`` markers are a past // record, not a tool-call syntax to emit. diff --git a/packages/ai/test/google-empty-response-retry.test.ts b/packages/ai/test/google-empty-response-retry.test.ts index 2e66440c1..70a0e1d67 100644 --- a/packages/ai/test/google-empty-response-retry.test.ts +++ b/packages/ai/test/google-empty-response-retry.test.ts @@ -81,6 +81,25 @@ const cliModel: Model<"google-gemini-cli"> = buildModel({ maxTokens: 32_000, }); +const ANTIGRAVITY_DAILY_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com"; +const ANTIGRAVITY_SANDBOX_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; + +const antigravityModel: Model<"google-gemini-cli"> = buildModel({ + ...cliModel, + provider: "google-antigravity", + baseUrl: ANTIGRAVITY_DAILY_ENDPOINT, +}); + +function withResponseUrl(response: Response, endpoint: string): Response { + Object.defineProperty(response, "url", { value: `${endpoint}/v1internal:streamGenerateContent?alt=sse` }); + return response; +} + +function endpointFromInput(input: Parameters[0]): string { + const url = input instanceof Request ? input.url : input.toString(); + return url.startsWith(ANTIGRAVITY_SANDBOX_ENDPOINT) ? ANTIGRAVITY_SANDBOX_ENDPOINT : ANTIGRAVITY_DAILY_ENDPOINT; +} + describe("Google empty-response retry (public + Vertex path)", () => { it("retries a STOP-with-empty-text response and delivers the real follow-up content", async () => { let calls = 0; @@ -235,6 +254,207 @@ describe("Google empty-response retry (Cloud Code Assist path)", () => { void events; }); + it("retries after discarding a planning leak and delivers one structured function call", async () => { + let calls = 0; + const fetchMock: FetchImpl = async () => { + calls += 1; + const response = + calls === 1 + ? sse(ccaChunk('{\n "thought": "inspect the project",\n "call": "lookup"\n}')) + : sse({ + response: { + candidates: [ + { + content: { + parts: [{ functionCall: { name: "lookup", args: { q: "x" }, id: "call_1" } }], + }, + finishReason: "STOP", + }, + ], + }, + }); + Object.defineProperty(response, "url", { value: "https://example.com/v1internal:streamGenerateContent" }); + return response; + }; + + const stream = streamGoogleGeminiCli(cliModel, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + fetch: fetchMock, + }); + const { events, starts } = await drain(stream); + const result = await stream.result(); + + expect(calls).toBe(2); + expect(starts).toBe(1); + expect(result.stopReason).toBe("toolUse"); + expect(result.content).toHaveLength(1); + expect(result.content[0]).toMatchObject({ + type: "toolCall", + id: "call_1", + name: "lookup", + arguments: { q: "x" }, + }); + expect(events.filter(e => e.type === "toolcall_start")).toHaveLength(1); + }); + + it("fails over to the sandbox endpoint after daily returns only empty successful streams", async () => { + const requestedEndpoints: string[] = []; + const fetchMock: FetchImpl = async input => { + const endpoint = endpointFromInput(input); + requestedEndpoints.push(endpoint); + const response = endpoint === ANTIGRAVITY_SANDBOX_ENDPOINT ? sse(ccaChunk("Recovered.")) : sse(ccaChunk("")); + return withResponseUrl(response, endpoint); + }; + + const stream = streamGoogleGeminiCli(antigravityModel, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + antigravityEndpointMode: "auto", + fetch: fetchMock, + }); + const { starts } = await drain(stream); + const result = await stream.result(); + + expect(requestedEndpoints).toEqual([ + ANTIGRAVITY_DAILY_ENDPOINT, + ANTIGRAVITY_DAILY_ENDPOINT, + ANTIGRAVITY_DAILY_ENDPOINT, + ANTIGRAVITY_SANDBOX_ENDPOINT, + ]); + expect(starts).toBe(1); + expect(result.stopReason).toBe("stop"); + expect(textOf(result)).toBe("Recovered."); + }); + + for (const { mode, endpoint } of [ + { mode: "production", endpoint: ANTIGRAVITY_DAILY_ENDPOINT }, + { mode: "sandbox", endpoint: ANTIGRAVITY_SANDBOX_ENDPOINT }, + ] as const) { + it(`keeps empty-response retries on the selected ${mode} endpoint`, async () => { + const requestedEndpoints: string[] = []; + const fetchMock: FetchImpl = async input => { + const requestedEndpoint = endpointFromInput(input); + requestedEndpoints.push(requestedEndpoint); + return withResponseUrl(sse(ccaChunk("")), requestedEndpoint); + }; + + const stream = streamGoogleGeminiCli(antigravityModel, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + antigravityEndpointMode: mode, + fetch: fetchMock, + }); + const result = await stream.result(); + + expect(requestedEndpoints).toEqual([endpoint, endpoint, endpoint]); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("empty response"); + }); + } + + for (const { name, chunk, errorText } of [ + { + name: "SAFETY finish", + chunk: { response: { candidates: [{ content: { parts: [] }, finishReason: "SAFETY" }] } }, + errorText: "SAFETY", + }, + { + name: "MALFORMED_FUNCTION_CALL finish", + chunk: { + response: { candidates: [{ content: { parts: [] }, finishReason: "MALFORMED_FUNCTION_CALL" }] }, + }, + errorText: "MALFORMED_FUNCTION_CALL", + }, + { + name: "PROHIBITED_CONTENT block", + chunk: { + response: { + candidates: [], + promptFeedback: { blockReason: "PROHIBITED_CONTENT", blockReasonMessage: "policy blocked" }, + }, + }, + errorText: "PROHIBITED_CONTENT", + }, + ] as const) { + it(`does not fail over after a terminal ${name}`, async () => { + const requestedEndpoints: string[] = []; + const fetchMock: FetchImpl = async input => { + const endpoint = endpointFromInput(input); + requestedEndpoints.push(endpoint); + return withResponseUrl(sse(chunk), endpoint); + }; + + const stream = streamGoogleGeminiCli(antigravityModel, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + antigravityEndpointMode: "auto", + fetch: fetchMock, + }); + const result = await stream.result(); + + expect(requestedEndpoints).toEqual([ANTIGRAVITY_DAILY_ENDPOINT]); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain(errorText); + }); + } + + it("does not fail over after an account-verification HTTP 403", async () => { + const requestedEndpoints: string[] = []; + const fetchMock: FetchImpl = async input => { + const endpoint = endpointFromInput(input); + requestedEndpoints.push(endpoint); + return new Response( + JSON.stringify({ + error: { + code: 403, + status: "PERMISSION_DENIED", + details: [ + { + "@type": "type.googleapis.com/google.rpc.ErrorInfo", + reason: "VALIDATION_REQUIRED", + metadata: { validation_url: "https://accounts.google.com/verify" }, + }, + ], + }, + }), + { status: 403 }, + ); + }; + + const stream = streamGoogleGeminiCli(antigravityModel, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + antigravityEndpointMode: "auto", + fetch: fetchMock, + }); + const result = await stream.result(); + + expect(requestedEndpoints).toEqual([ANTIGRAVITY_DAILY_ENDPOINT]); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("Account verification required"); + }); + + it("does not fail over after partial output has started", async () => { + const requestedEndpoints: string[] = []; + const fetchMock: FetchImpl = async input => { + const endpoint = endpointFromInput(input); + requestedEndpoints.push(endpoint); + return withResponseUrl( + sse({ response: { candidates: [{ content: { parts: [{ text: "partial" }] } }] } }), + endpoint, + ); + }; + + const stream = streamGoogleGeminiCli(antigravityModel, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + antigravityEndpointMode: "auto", + fetch: fetchMock, + }); + const { starts } = await drain(stream); + const result = await stream.result(); + + expect(requestedEndpoints).toEqual([ANTIGRAVITY_DAILY_ENDPOINT]); + expect(starts).toBe(1); + expect(result.stopReason).toBe("error"); + expect(textOf(result)).toBe("partial"); + }); + it("does not coalesce function-call thought signatures into the preceding text block", async () => { const chunks = [ { response: { candidates: [{ content: { parts: [{ text: "Done" }] } }] } }, diff --git a/packages/ai/test/google-gemini-cli-alignment.test.ts b/packages/ai/test/google-gemini-cli-alignment.test.ts index 3e8c2f463..baa4f3b82 100644 --- a/packages/ai/test/google-gemini-cli-alignment.test.ts +++ b/packages/ai/test/google-gemini-cli-alignment.test.ts @@ -557,18 +557,26 @@ describe("Google Gemini CLI alignment", () => { }); describe("planning leak interception", () => { - it("intercepts fragmented planning leak and discards it", async () => { + it("intercepts a fragmented planning leak and retries after discarding it", async () => { const sseChunks = [ 'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":"{\\n"}]}}]}}\n\n', 'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":" \\"thought\\": \\"let us do something\\",\\n"}]}}]}}\n\n', 'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":" \\"call\\": \\"read\\",\\n \\"paths\\": [\\"src/main.ts\\"]\\n}"}]},"finishReason":"STOP"}]}}\n\n', ]; + let fetchCalls = 0; const fetchMock: FetchImpl = async () => { + fetchCalls += 1; + const chunks = + fetchCalls === 1 + ? sseChunks + : [ + 'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":"Recovered."}]},"finishReason":"STOP"}]}}\n\n', + ]; const stream = new ReadableStream({ async start(controller) { const encoder = new TextEncoder(); - for (const chunk of sseChunks) { + for (const chunk of chunks) { controller.enqueue(encoder.encode(chunk)); await Bun.sleep(5); } @@ -596,13 +604,13 @@ describe("Google Gemini CLI alignment", () => { } const result = await stream.result(); - // A fully-discarded planning leak leaves no residual content — no empty - // text block survives (the central healing wrapper strips empties too). - expect(result.content).toHaveLength(0); + expect(fetchCalls).toBe(2); + expect(result.content).toEqual([{ type: "text", text: "Recovered." }]); expect(result.stopReason).toBe("stop"); const textDeltaEvents = events.filter(e => e.type === "text_delta"); - expect(textDeltaEvents).toHaveLength(0); + expect(textDeltaEvents).toHaveLength(1); + expect(textDeltaEvents[0].delta).toBe("Recovered."); }); it("does not intercept normal JSON starting with { and releases it", async () => { diff --git a/packages/ai/test/issue-2883-moonshot-base-url.test.ts b/packages/ai/test/issue-2883-moonshot-base-url.test.ts index 66fe160c0..93934111d 100644 --- a/packages/ai/test/issue-2883-moonshot-base-url.test.ts +++ b/packages/ai/test/issue-2883-moonshot-base-url.test.ts @@ -1,5 +1,7 @@ -import { afterEach, describe, expect, test } from "bun:test"; +import { afterEach, describe, expect, test, vi } from "bun:test"; import { resolveOpenAIRequestSetup } from "@oh-my-pi/pi-ai/providers/openai-shared"; +import { loginMoonshot } from "@oh-my-pi/pi-ai/registry/moonshot"; +import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; const ORIGINAL_MOONSHOT_BASE_URL = Bun.env.MOONSHOT_BASE_URL; @@ -42,6 +44,23 @@ describe("Moonshot China base URL override (issue #2883)", () => { expect(setup.baseUrl).toBe("https://api.moonshot.ai/v1"); }); + test("validates login against the configured Moonshot endpoint", async () => { + Bun.env.MOONSHOT_BASE_URL = "https://api.moonshot.cn/v1/"; + const fetchMock: FetchImpl = vi.fn(async (input: string | URL | Request) => { + const url = typeof input === "string" ? input : input.toString(); + expect(url).toBe("https://api.moonshot.cn/v1/models"); + return new Response(JSON.stringify({ object: "list", data: [] }), { status: 200 }); + }); + + const apiKey = await loginMoonshot({ + onPrompt: async () => " sk-china-key ", + fetch: fetchMock, + }); + + expect(apiKey).toBe("sk-china-key"); + expect(fetchMock).toHaveBeenCalledTimes(1); + }); + test("does not redirect other openai-completions providers", () => { Bun.env.MOONSHOT_BASE_URL = "https://api.moonshot.cn/v1"; const setup = resolveOpenAIRequestSetup( diff --git a/packages/ai/test/issue-5983-repro.test.ts b/packages/ai/test/issue-5983-repro.test.ts new file mode 100644 index 000000000..3aaf888f7 --- /dev/null +++ b/packages/ai/test/issue-5983-repro.test.ts @@ -0,0 +1,76 @@ +import { describe, expect, it } from "bun:test"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; + +const model = buildModel({ + id: "kimi-k3", + name: "Kimi K3", + api: "openai-completions", + provider: "litellm", + baseUrl: "http://127.0.0.1:4000/v1", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 262_144, + maxTokens: 32_768, +}); + +const context: Context = { + messages: [{ role: "user", content: "hello", timestamp: 0 }], +}; + +async function capturePayload(reasoning: Effort): Promise { + let payload: unknown; + const fetchMock: FetchImpl = Object.assign( + async (_input: string | URL | Request, init?: RequestInit): Promise => { + payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}"); + return new Response( + 'data: {"id":"x","object":"chat.completion.chunk","created":0,"model":"kimi-k3","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', + { headers: { "content-type": "text/event-stream" } }, + ); + }, + { preconnect: fetch.preconnect }, + ); + + await streamOpenAICompletions(model, context, { + apiKey: "test-key", + fetch: fetchMock, + reasoning, + }).result(); + return payload; +} + +describe("issue #5983 — Kimi K3 OpenAI-compatible effort contract", () => { + it("derives K3's mandatory low/high/max ladder for a LiteLLM route", () => { + expect(getSupportedEfforts(model)).toEqual([Effort.Low, Effort.High, Effort.Max]); + expect(model.thinking).toMatchObject({ + mode: "effort", + defaultLevel: Effort.Max, + requiresEffort: true, + }); + expect(model.compat.reasoningEffortMap).toEqual({ + minimal: "low", + medium: "high", + xhigh: "max", + max: "max", + }); + }); + + it("folds every generic requested tier into K3's accepted wire values", async () => { + const cases: readonly (readonly [Effort, string])[] = [ + [Effort.Minimal, "low"], + [Effort.Low, "low"], + [Effort.Medium, "high"], + [Effort.High, "high"], + [Effort.XHigh, "max"], + [Effort.Max, "max"], + ]; + const payloads = await Promise.all(cases.map(([requested]) => capturePayload(requested))); + for (let index = 0; index < cases.length; index++) { + expect(payloads[index]).toMatchObject({ reasoning_effort: cases[index]?.[1] }); + } + }); +}); diff --git a/packages/ai/test/issue-967-vision-guard.test.ts b/packages/ai/test/issue-967-vision-guard.test.ts index 08e9ef47c..679e5741e 100644 --- a/packages/ai/test/issue-967-vision-guard.test.ts +++ b/packages/ai/test/issue-967-vision-guard.test.ts @@ -27,6 +27,7 @@ const compat: ResolvedOpenAICompat = { supportsMultipleSystemMessages: true, supportsReasoningEffort: true, supportsReasoningParams: true, + supportsSamplingParams: true, alwaysSendMaxTokens: false, isOpenRouterHost: false, isVercelGatewayHost: false, diff --git a/packages/ai/test/kimi-usage.test.ts b/packages/ai/test/kimi-usage.test.ts new file mode 100644 index 000000000..b86b66310 --- /dev/null +++ b/packages/ai/test/kimi-usage.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it } from "bun:test"; +import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; +import type { UsageFetchContext, UsageFetchParams } from "@oh-my-pi/pi-ai/usage"; +import { kimiUsageProvider } from "@oh-my-pi/pi-ai/usage/kimi"; + +function makeCredential(): UsageFetchParams["credential"] { + return { + type: "oauth", + accessToken: "kimi-test-token", + }; +} + +function makeCtx(payload: unknown): UsageFetchContext { + const fetch: FetchImpl = async () => + new Response(JSON.stringify(payload), { + status: 200, + headers: { "content-type": "application/json" }, + }); + return { fetch }; +} + +describe("kimi usage provider", () => { + it("surfaces the 5h limit reset time from the limit detail onto the window", async () => { + // Live payload shape: `resetTime` lives on `detail`, while `window` + // carries only duration/timeUnit. The 5h row must still render + // "resets in …" in `omp usage`. + const detailReset = "2026-07-18T05:43:35.355947Z"; + const usageReset = "2026-07-21T07:43:35.355947Z"; + const report = await kimiUsageProvider.fetchUsage!( + { provider: "kimi-code", credential: makeCredential(), signal: undefined }, + makeCtx({ + usage: { limit: "100", used: "28", remaining: "72", resetTime: usageReset }, + limits: [ + { + window: { duration: 300, timeUnit: "TIME_UNIT_MINUTE" }, + detail: { limit: "100", remaining: "100", resetTime: detailReset }, + }, + ], + }), + ); + + expect(report).not.toBeNull(); + expect(report!.limits).toHaveLength(2); + + const total = report!.limits[0]!; + expect(total.label).toBe("Total quota"); + expect(total.window?.resetsAt).toBe(Date.parse(usageReset)); + + const fiveHour = report!.limits[1]!; + expect(fiveHour.label).toBe("5h limit"); + expect(fiveHour.window?.durationMs).toBe(5 * 60 * 60 * 1000); + expect(fiveHour.window?.resetsAt).toBe(Date.parse(detailReset)); + }); + + it("keeps an explicit window resetTime authoritative over the detail one", async () => { + const windowReset = "2026-07-18T06:00:00.000Z"; + const detailReset = "2026-07-18T05:43:35.355947Z"; + const report = await kimiUsageProvider.fetchUsage!( + { provider: "kimi-code", credential: makeCredential(), signal: undefined }, + makeCtx({ + limits: [ + { + window: { duration: 300, timeUnit: "TIME_UNIT_MINUTE", resetTime: windowReset }, + detail: { limit: "100", remaining: "40", resetTime: detailReset }, + }, + ], + }), + ); + + expect(report).not.toBeNull(); + expect(report!.limits).toHaveLength(1); + expect(report!.limits[0]!.window?.resetsAt).toBe(Date.parse(windowReset)); + }); +}); diff --git a/packages/ai/test/leaked-thinking-stream.test.ts b/packages/ai/test/leaked-thinking-stream.test.ts index fadd62c0f..98dea52fa 100644 --- a/packages/ai/test/leaked-thinking-stream.test.ts +++ b/packages/ai/test/leaked-thinking-stream.test.ts @@ -224,6 +224,88 @@ describe("wrapLeakedThinkingStream", () => { expect(calls[0]?.thoughtSignature).toBe("tsig"); }); + it("captures a native thinking signature delivered at thinking_end, not on the delta", async () => { + // Anthropic sends thinking_delta first (no signature), then signature_delta, + // so the completed signature only appears on the thinking_end partial. + const signature = `SIG_${"x".repeat(400)}`; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "thinking_delta", + contentIndex: 0, + delta: "reason", + partial: msg({ content: [{ type: "thinking", thinking: "reason", thinkingSignature: "" }] }), + }); + const signed = msg({ content: [{ type: "thinking", thinking: "reason", thinkingSignature: signature }] }); + inner.push({ type: "thinking_end", contentIndex: 0, content: "reason", partial: signed }); + inner.push({ type: "done", reason: "stop", message: signed }); + }); + + expect(thinks(result).map(b => b.thinkingSignature)).toEqual([signature]); + }); + + it("emits a late native thinking signature by source index after later blocks start", async () => { + const firstSignature = "sig-first"; + const secondSignature = "sig-second"; + const call: ToolCall = { + type: "toolCall", + id: "call_between_thinking", + name: "read", + arguments: { path: "x" }, + }; + const first = { type: "thinking" as const, thinking: "first", thinkingSignature: "" }; + const second = { type: "thinking" as const, thinking: "second", thinkingSignature: "" }; + const { events, result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "thinking_delta", + contentIndex: 0, + delta: first.thinking, + partial: msg({ content: [first] }), + }); + const withCall = msg({ content: [first, call] }); + inner.push({ type: "toolcall_start", contentIndex: 1, partial: withCall }); + inner.push({ type: "toolcall_end", contentIndex: 1, toolCall: call, partial: withCall }); + inner.push({ + type: "thinking_delta", + contentIndex: 2, + delta: second.thinking, + partial: msg({ content: [first, call, second] }), + }); + const firstSigned = { + ...first, + thinkingSignature: firstSignature, + }; + const secondSigned = { + ...second, + thinkingSignature: secondSignature, + }; + inner.push({ + type: "thinking_end", + contentIndex: 0, + content: first.thinking, + partial: msg({ content: [firstSigned, call, second] }), + }); + const signed = msg({ content: [firstSigned, call, secondSigned] }); + inner.push({ type: "thinking_end", contentIndex: 2, content: second.thinking, partial: signed }); + inner.push({ type: "done", reason: "stop", message: signed }); + }); + + const firstEnd = events.find(event => event.type === "thinking_end" && event.contentIndex === 0); + if (firstEnd?.type !== "thinking_end") throw new Error("Missing first projected thinking_end"); + expect(firstEnd.partial.content[firstEnd.contentIndex]).toEqual({ + type: "thinking", + thinking: first.thinking, + thinkingSignature: firstSignature, + }); + expect(events.indexOf(firstEnd)).toBeGreaterThan(events.findIndex(event => event.type === "toolcall_start")); + expect(events.filter(event => event.type === "thinking_end" && event.contentIndex === 0)).toHaveLength(1); + expect(thinks(result).map(block => [block.thinking, block.thinkingSignature])).toEqual([ + ["first", firstSignature], + ["second", secondSignature], + ]); + }); + it("preserves native tool-call ids and streamed partial JSON while healing", async () => { const inner = new AssistantMessageEventStream(); const out = wrapLeakedThinkingStream(inner); diff --git a/packages/ai/test/model-cache.test.ts b/packages/ai/test/model-cache.test.ts index 2dab26910..e520c4f1e 100644 --- a/packages/ai/test/model-cache.test.ts +++ b/packages/ai/test/model-cache.test.ts @@ -47,8 +47,11 @@ describe("model cache migrations", () => { } }); - it("invalidates legacy cached models and lets the next discovery write fresh ones", () => { - const legacyModel = createModel("legacy-cloud-model", "Legacy Cloud Model"); + it("invalidates and scrubs pre-v10 header-bearing cache rows", async () => { + const legacyModel = { + ...createModel("legacy-cloud-model", "Legacy Cloud Model"), + headers: { "X-Access-Token": "legacy-cached-secret" }, + }; const legacyDb = new Database(dbPath, { create: true }); legacyDb.run(` CREATE TABLE model_cache ( @@ -61,12 +64,13 @@ describe("model cache migrations", () => { `); legacyDb.run( "INSERT INTO model_cache (provider_id, version, updated_at, authoritative, models) VALUES (?, ?, ?, ?, ?)", - ["ollama-cloud", 2, Date.now(), 1, JSON.stringify([legacyModel])], + ["ollama-cloud", 9, Date.now(), 1, JSON.stringify([legacyModel])], ); legacyDb.close(); const migrated = readModelCache<"openai-completions">("ollama-cloud", TTL_MS, Date.now, dbPath); expect(migrated).toBeNull(); + expect((await fs.readFile(dbPath)).includes("legacy-cached-secret")).toBe(false); const replacementModel = createModel("fresh-cloud-model", "Fresh Cloud Model"); writeModelCache("ollama-cloud", Date.now(), [replacementModel], true, "static-v3", dbPath); @@ -75,4 +79,43 @@ describe("model cache migrations", () => { expect(fresh?.models.map(model => model.id)).toEqual(["fresh-cloud-model"]); expect(fresh?.staticFingerprint).toBe("static-v3"); }); + + it("omits every model header before persisting (#5780)", () => { + const model = buildModel({ + id: "gated-model", + name: "Gated Model", + api: "openai-completions", + provider: "runtime-ext", + baseUrl: "https://ext.example.com/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 4096, + maxTokens: 1024, + headers: { + Authorization: "Bearer standard-secret", + "X-Goog-Api-Key": "google-secret", + "X-Access-Token": "access-secret", + "X-Project-Id": "proj-42", + }, + }); + writeModelCache("runtime-ext", Date.now(), [model], true, "static-v1", dbPath); + + // Header names are provider-defined and any value may be a credential. + // The plaintext SQLite payload therefore persists no model headers. + const raw = new Database(dbPath, { readonly: true }); + const row = raw + .query<{ models: string }, []>("SELECT models FROM model_cache WHERE provider_id = 'runtime-ext'") + .get(); + raw.close(); + expect(row?.models).not.toContain("standard-secret"); + expect(row?.models).not.toContain("google-secret"); + expect(row?.models).not.toContain("access-secret"); + expect(row?.models).not.toContain("proj-42"); + + const cached = readModelCache<"openai-completions">("runtime-ext", TTL_MS, Date.now, dbPath); + expect(cached?.models[0]?.headers).toBeUndefined(); + expect(cached?.headerOmittedModelIds).toEqual(["gated-model"]); + expect(cached?.unrestorableHeaderModelIds).toEqual(["gated-model"]); + }); }); diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 69d1d8c2b..b650eb481 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -187,25 +187,24 @@ describe("openai-codex reasoning.context", () => { // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns` // ("Unsupported value: 'all_turns' is not supported with this model"). - it.each([ - "gpt-5.1-codex", - "gpt-5.3-codex", - "gpt-5.3-codex-spark", - ])("omits the all_turns default for pre-5.4 model %s", async modelId => { - const model = createCodexModel(modelId); + it.each(["gpt-5.1-codex", "gpt-5.3-codex", "gpt-5.3-codex-spark"])( + "omits the all_turns default for pre-5.4 model %s", + async modelId => { + const model = createCodexModel(modelId); - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning).toBeDefined(); - expect(defaulted.reasoning?.context).toBeUndefined(); - expect("context" in (defaulted.reasoning ?? {})).toBe(false); + const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); + expect(defaulted.reasoning).toBeDefined(); + expect(defaulted.reasoning?.context).toBeUndefined(); + expect("context" in (defaulted.reasoning ?? {})).toBe(false); - // A supported override (current_turn/auto) is still honored. - const overridden = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "medium", - reasoningContext: "current_turn", - }); - expect(overridden.reasoning?.context).toBe("current_turn"); - }); + // A supported override (current_turn/auto) is still honored. + const overridden = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningContext: "current_turn", + }); + expect(overridden.reasoning?.context).toBe("current_turn"); + }, + ); it("suppresses an explicit all_turns override on a pre-5.4 model", async () => { const model = createCodexModel("gpt-5.3-codex-spark"); @@ -241,24 +240,23 @@ describe("openai-codex reasoning.summary", () => { // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `reasoning.summary` // ("Unsupported parameter: 'reasoning.summary' is not supported with this model"). - it.each([ - "gpt-5.1-codex", - "gpt-5.3-codex", - "gpt-5.3-codex-spark", - ])("omits reasoning.summary for pre-5.4 model %s", async modelId => { - const model = createCodexModel(modelId); + it.each(["gpt-5.1-codex", "gpt-5.3-codex", "gpt-5.3-codex-spark"])( + "omits reasoning.summary for pre-5.4 model %s", + async modelId => { + const model = createCodexModel(modelId); - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning).toBeDefined(); - expect("summary" in (defaulted.reasoning ?? {})).toBe(false); + const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); + expect(defaulted.reasoning).toBeDefined(); + expect("summary" in (defaulted.reasoning ?? {})).toBe(false); - // Even an explicit summary level is suppressed on unsupported ids. - const forced = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "medium", - reasoningSummary: "detailed", - }); - expect("summary" in (forced.reasoning ?? {})).toBe(false); - }); + // Even an explicit summary level is suppressed on unsupported ids. + const forced = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningSummary: "detailed", + }); + expect("summary" in (forced.reasoning ?? {})).toBe(false); + }, + ); }); describe("openai-codex Responses Lite input shaping", () => { @@ -344,6 +342,25 @@ describe("openai-codex Responses Lite input shaping", () => { expect(noTools.parallel_tool_calls).toBe(false); }); + it("falls back from forced hosted tool choices without weakening explicit tool-use constraints", async () => { + const model = createCodexModel("gpt-5.6-terra"); + const tools = [{ type: "function", name: "handoff", parameters: { type: "object" } }]; + + const forced = await transformRequestBody( + { model: model.id, tools, tool_choice: { type: "web_search" } }, + model, + { responsesLite: true }, + ); + expect(forced.tool_choice).toBe("auto"); + expect(forced.tools).toBeUndefined(); + + const disabled = await transformRequestBody({ model: model.id, tools, tool_choice: "none" }, model, { + responsesLite: true, + }); + expect(disabled.tool_choice).toBe("none"); + expect(disabled.tools).toBeUndefined(); + }); + it("moves instructions and tools into input items under lite", async () => { const model = createCodexModel("gpt-5.6-terra"); const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }]; @@ -1186,3 +1203,34 @@ describe("openai-codex concurrent reasoning summaries", () => { expect(text?.text).toBe("Hello"); }); }); + +describe("openai-codex native history redaction", () => { + it("redacts credentials from user provider history before replaying it", () => { + const model = createCodexModel("gpt-5.1-codex"); + const credential = "sk-ABCdef1234567890ABCdef1234567890ABCdef1234567890ABCdef123456"; + const context: Context = { + messages: [ + { + role: "user", + content: "fallback", + timestamp: Date.now(), + providerPayload: { + type: "openaiResponsesHistory", + provider: model.provider, + items: [{ type: "message", role: "user", content: [{ type: "input_text", text: credential }] }], + }, + } as Context["messages"][number], + ], + }; + + const messages = convertCodexResponsesMessages(model, context); + + expect(messages).toEqual([ + { + type: "message", + role: "user", + content: [{ type: "input_text", text: "[openai_token_redacted]" }], + }, + ]); + }); +}); diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 4d8059fdf..56699f1e0 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -16,6 +16,7 @@ import type { ModelSpec, ProviderSessionState, } from "@oh-my-pi/pi-ai/types"; +import { __resetProxyCache } from "@oh-my-pi/pi-ai/utils/proxy"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import * as piUtils from "@oh-my-pi/pi-utils"; @@ -24,6 +25,16 @@ const { getAgentDir, setAgentDir, TempDir } = piUtils; const originalAgentDir = getAgentDir(); const originalWebSocket = global.WebSocket; const originalCodexWebSocketV2 = Bun.env.PI_CODEX_WEBSOCKET_V2; +const originalProxyEnv: Record = { + PI_PROXY: Bun.env.PI_PROXY, + PI_PROXY_CODEX_PROXY_TEST: Bun.env.PI_PROXY_CODEX_PROXY_TEST, + HTTPS_PROXY: Bun.env.HTTPS_PROXY, + https_proxy: Bun.env.https_proxy, + ALL_PROXY: Bun.env.ALL_PROXY, + all_proxy: Bun.env.all_proxy, + NO_PROXY: Bun.env.NO_PROXY, + no_proxy: Bun.env.no_proxy, +}; const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; function restoreEnv(name: string, value: string | undefined): void { @@ -35,6 +46,8 @@ function restoreEnv(name: string, value: string | undefined): void { } beforeEach(() => { + for (const key in originalProxyEnv) delete Bun.env[key]; + __resetProxyCache(); vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); }); @@ -43,6 +56,8 @@ afterEach(() => { setAgentDir(originalAgentDir); restoreEnv("PI_CODEX_WEBSOCKET_V2", originalCodexWebSocketV2); vi.useRealTimers(); + for (const key in originalProxyEnv) restoreEnv(key, originalProxyEnv[key]); + __resetProxyCache(); vi.restoreAllMocks(); }); @@ -171,6 +186,7 @@ function encodeWebSocketMessage(value: Record): Uint8Array { } type WsHeaders = Record; +type WsOptions = { headers?: WsHeaders; proxy?: string }; type WsEventType = "open" | "message" | "error" | "close"; type CodexTestUsage = { @@ -210,7 +226,7 @@ class MockWebSocket { constructor( public readonly url: string, - public readonly options?: { headers?: WsHeaders }, + public readonly options?: WsOptions, ) {} send(_data: string): void {} @@ -1378,6 +1394,107 @@ describe("openai-codex streaming", () => { expect(Object.keys(capturedHeaders ?? {}).filter(key => key.toLowerCase() === "openai-beta")).toHaveLength(1); }); + it("passes the provider proxy to websocket handshakes", async () => { + const proxy = "socks5://127.0.0.1:7890"; + Bun.env.PI_PROXY_CODEX_PROXY_TEST = proxy; + __resetProxyCache(); + let capturedProxy: string | undefined; + class ProxyCaptureWebSocket extends MockWebSocket { + constructor(url: string, options?: WsOptions) { + super(url, options); + capturedProxy = options?.proxy; + this.scheduleOpen(); + } + } + global.WebSocket = ProxyCaptureWebSocket as unknown as typeof WebSocket; + const model = { + ...createCodexTestModel("https://chatgpt.com/backend-api"), + provider: "codex-proxy-test", + }; + const providerSessionState = new Map(); + + try { + await prewarmOpenAICodexResponses(model, { + apiKey: createCodexTestToken(), + sessionId: "ws-proxy-session", + providerSessionState, + }); + expect(capturedProxy).toBe(proxy); + } finally { + for (const state of providerSessionState.values()) state.close(); + delete Bun.env.PI_PROXY_CODEX_PROXY_TEST; + } + }); + + it("falls back to standard proxy variables for websocket handshakes", async () => { + const cases: Array<{ env: string; proxy: string }> = [ + { env: "HTTPS_PROXY", proxy: "http://127.0.0.1:7890" }, + { env: "ALL_PROXY", proxy: "socks5://127.0.0.1:7891" }, + ]; + + for (const { env, proxy } of cases) { + delete Bun.env.HTTPS_PROXY; + delete Bun.env.ALL_PROXY; + Bun.env[env] = proxy; + let capturedProxy: string | undefined; + class StandardProxyWebSocket extends MockWebSocket { + constructor(url: string, options?: WsOptions) { + super(url, options); + capturedProxy = options?.proxy; + this.scheduleOpen(); + } + } + global.WebSocket = StandardProxyWebSocket as unknown as typeof WebSocket; + const model = { + ...createCodexTestModel("https://chatgpt.com/backend-api"), + provider: `codex-${env.toLowerCase()}-test`, + }; + const providerSessionState = new Map(); + + try { + await prewarmOpenAICodexResponses(model, { + apiKey: createCodexTestToken(), + sessionId: `ws-${env.toLowerCase()}-proxy-session`, + providerSessionState, + }); + expect(capturedProxy).toBe(proxy); + } finally { + for (const state of providerSessionState.values()) state.close(); + } + } + }); + + it("bypasses configured proxies for NO_PROXY websocket targets", async () => { + Bun.env.PI_PROXY_CODEX_PROXY_TEST = "http://127.0.0.1:7890"; + Bun.env.NO_PROXY = "chatgpt.com:443"; + __resetProxyCache(); + let capturedProxy: string | undefined; + class NoProxyWebSocket extends MockWebSocket { + constructor(url: string, options?: WsOptions) { + super(url, options); + capturedProxy = options?.proxy; + this.scheduleOpen(); + } + } + global.WebSocket = NoProxyWebSocket as unknown as typeof WebSocket; + const model = { + ...createCodexTestModel("https://chatgpt.com/backend-api"), + provider: "codex-proxy-test", + }; + const providerSessionState = new Map(); + + try { + await prewarmOpenAICodexResponses(model, { + apiKey: createCodexTestToken(), + sessionId: "ws-no-proxy-session", + providerSessionState, + }); + expect(capturedProxy).toBeUndefined(); + } finally { + for (const state of providerSessionState.values()) state.close(); + } + }); + it("sends the Responses Lite marker on the upgrade and in response.create client_metadata", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 4c9d8ff0e..18dae3edc 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -115,7 +115,7 @@ function kimiZaiModel(): Model<"openai-completions"> { async function captureOpenAICompletionsPayload( model: Model<"openai-completions">, context: Context = baseContext(), - options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max" }, + options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; temperature?: number }, ): Promise { const { promise, resolve } = Promise.withResolvers(); const fetchMock = createMockFetch(["[DONE]"]); @@ -156,6 +156,19 @@ function getLastTextPart(content: unknown): Record | undefined } describe("openai-completions compatibility", () => { + it("omits sampling params for OpenAI reasoning models", async () => { + const model = buildModel({ + ...gpt4oMiniSpec, + id: "gpt-5.6-luna", + provider: "github-copilot", + api: "openai-completions", + } as ModelSpec<"openai-completions">); + expect(model.compat.supportsSamplingParams).toBe(false); + + const payload = await captureOpenAICompletionsPayload(model, undefined, { temperature: 0 }); + expect(toObject(payload)?.temperature).toBeUndefined(); + }); + it("serializes assistant text content as a plain string", () => { const model: Model<"openai-completions"> = buildModel({ ...gpt4oMiniSpec, @@ -196,6 +209,7 @@ describe("openai-completions compatibility", () => { supportsStrictMode: true, toolStrictMode: "none", supportsReasoningParams: true, + supportsSamplingParams: true, alwaysSendMaxTokens: false, isOpenRouterHost: false, isVercelGatewayHost: false, @@ -2368,6 +2382,22 @@ describe("Moonshot Flavored JSON Schema tool normalization", () => { additionalProperties: false, }, }, + { + name: "task", + description: "spawn task", + parameters: { + type: "object", + properties: { + tasks: { + type: "array", + items: { + type: "object", + properties: { outputSchema: true }, + }, + }, + }, + }, + }, ]; function toolParameters(payload: unknown, toolName: string): Record { @@ -2405,8 +2435,8 @@ describe("Moonshot Flavored JSON Schema tool normalization", () => { return buildModel({ ...gpt4oMiniSpec, api: "openai-completions", - provider: "vllm", - baseUrl: "http://localhost:8000/v1", + provider: "custom", + baseUrl: "https://api.example.com/v1", id: "local-model", } as ModelSpec<"openai-completions">); } @@ -2420,6 +2450,10 @@ describe("Moonshot Flavored JSON Schema tool normalization", () => { const paths = probeProperty(payload, "find", "paths"); expect(paths.minItems).toBeUndefined(); expect(paths.type).toBe("array"); + const taskProperties = toObject( + toObject(toObject(probeProperty(payload, "task", "tasks").items)?.properties)?.outputSchema, + ); + expect(taskProperties).toEqual({}); }); it("leaves raw JSON Schema untouched on non-Moonshot hosts (flag-gated)", async () => { @@ -2432,5 +2466,136 @@ describe("Moonshot Flavored JSON Schema tool normalization", () => { expect(op).toEqual({ type: "string", enum: ["pr_checkout", "pr_create"], description: "github operation" }); const paths = probeProperty(payload, "find", "paths"); expect(paths.minItems).toBe(1); + const taskItems = toObject(probeProperty(payload, "task", "tasks").items); + const taskProperties = toObject(taskItems?.properties); + expect(taskProperties?.outputSchema).toBe(true); + }); +}); + +describe("grammar tool-schema normalization (issue #5914)", () => { + const primitiveUnion = { + anyOf: [ + { type: "string" }, + { type: "number" }, + { type: "boolean" }, + { type: "object" }, + { type: "array" }, + { type: "null" }, + ], + }; + + // An open field (`z.unknown()` / ArkType `"unknown"` / raw `{}`) becomes a + // bare boolean `true` after `toolWireSchema`'s empty-schema normalization + // (issue #1179). The `task` tool ships exactly this via `outputSchema`. + const openFieldTool: Tool = { + name: "task", + description: "spawn subagents", + parameters: { + type: "object", + properties: { + task: { type: "string" }, + outputSchema: {}, + nested: { + type: "object", + properties: { value: { type: "string" } }, + additionalProperties: false, + }, + }, + required: ["task"], + additionalProperties: false, + }, + }; + + function toolParameters(payload: unknown, toolName: string): Record { + const tools = toObject(payload)?.tools; + if (!Array.isArray(tools)) throw new Error("payload tools missing"); + for (const entry of tools) { + const fn = getNestedObject(entry, "function"); + if (fn?.name === toolName) { + const params = toObject(fn.parameters); + if (!params) throw new Error(`tool ${toolName} has no parameters`); + return params; + } + } + throw new Error(`tool ${toolName} not in payload`); + } + + function localLlamaModel(): Model<"openai-completions"> { + return buildModel({ + ...gpt4oMiniSpec, + api: "openai-completions", + provider: "llama.cpp", + baseUrl: "http://127.0.0.1:8080/v1", + id: "qwen3-coder", + } as ModelSpec<"openai-completions">); + } + + function remoteModel(): Model<"openai-completions"> { + return buildModel({ + ...gpt4oMiniSpec, + api: "openai-completions", + provider: "custom", + baseUrl: "https://api.example.com/v1", + id: "remote-model", + } as ModelSpec<"openai-completions">); + } + + it("auto-detects the grammar flavor for local OpenAI-compatible backends", () => { + expect(localLlamaModel().compat.toolSchemaFlavor).toBe("grammar"); + }); + + it("widens bare boolean subschemas and keeps additionalProperties:false", async () => { + const model = localLlamaModel(); + const payload = await captureOpenAICompletionsPayload(model, { ...baseContext(), tools: [openFieldTool] }); + const params = toolParameters(payload, "task"); + const properties = toObject(params.properties); + if (!properties) throw new Error("task tool has no properties"); + + // The offending bare `true` (from the `{}` open field) becomes a + // value-accepting primitive union the GBNF converter can compile. + expect(properties.outputSchema).toEqual(primitiveUnion); + // No bare boolean subschema remains anywhere in the wire schema. + expect(JSON.stringify(params)).not.toContain('"outputSchema":true'); + // The closed-object contract survives: dropping `additionalProperties: + // false` would silently reopen the object to arbitrary keys. + expect(params.additionalProperties).toBe(false); + const nested = toObject(properties.nested); + expect(nested?.additionalProperties).toBe(false); + }); + + it("widens booleans nested inside object-valued additionalProperties", async () => { + const model = localLlamaModel(); + const mapTool: Tool = { + name: "map_tool", + description: "tool with a schema-valued additionalProperties", + parameters: { + type: "object", + properties: { + env: { + type: "object", + additionalProperties: { type: "object", properties: { metadata: {} } }, + }, + }, + additionalProperties: false, + }, + }; + const payload = await captureOpenAICompletionsPayload(model, { ...baseContext(), tools: [mapTool] }); + const params = toolParameters(payload, "map_tool"); + const env = getNestedObject(toObject(params.properties), "env"); + const ap = toObject(env?.additionalProperties); + // The schema-valued form is traversed: the nested open field is widened + // so the GBNF converter never reaches a bare boolean subschema. + expect(getNestedObject(toObject(ap?.properties), "metadata")).toEqual(primitiveUnion); + // The boolean form on the outer object still survives untouched. + expect(params.additionalProperties).toBe(false); + }); + + it("leaves the wire schema untouched on non-grammar hosts", async () => { + const model = remoteModel(); + expect(model.compat.toolSchemaFlavor).toBeUndefined(); + const payload = await captureOpenAICompletionsPayload(model, { ...baseContext(), tools: [openFieldTool] }); + const properties = toObject(toolParameters(payload, "task").properties); + // Off the grammar path the open field keeps the normalized bare boolean. + expect(properties?.outputSchema).toBe(true); }); }); diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts index fe2731ecf..e598f6068 100644 --- a/packages/ai/test/openai-completions-tool-result-images.test.ts +++ b/packages/ai/test/openai-completions-tool-result-images.test.ts @@ -50,6 +50,7 @@ const compat: ResolvedOpenAICompat = { supportsStrictMode: true, toolStrictMode: "none", supportsReasoningParams: true, + supportsSamplingParams: true, alwaysSendMaxTokens: false, isOpenRouterHost: false, isVercelGatewayHost: false, diff --git a/packages/ai/test/openai-responses-history-payload.test.ts b/packages/ai/test/openai-responses-history-payload.test.ts index 78886cc84..583455d78 100644 --- a/packages/ai/test/openai-responses-history-payload.test.ts +++ b/packages/ai/test/openai-responses-history-payload.test.ts @@ -108,7 +108,7 @@ const assistantSnapshotContext: Context = { const codexAssistantSnapshotContext: Context = { messages: [ { role: "user", content: "generic history that should be replaced", timestamp: Date.now() }, - makeAssistantMessage(snapshotHistoryItems, false, "openai-codex", "gpt-5.2-codex"), + makeAssistantMessage(snapshotHistoryItems, false, "openai-codex", "gpt-5.5"), { role: "user", content: "follow-up user", timestamp: Date.now() }, ], }; @@ -117,7 +117,7 @@ const codexToCopilotContext: Context = { messages: [ { role: "user", content: "generic user before switch", timestamp: Date.now() }, { - ...makeAssistantMessage([], false, "openai-codex", "gpt-5.2-codex"), + ...makeAssistantMessage([], false, "openai-codex", "gpt-5.5"), content: [{ type: "text", text: "generic assistant that should be rebuilt" }], providerPayload: createOpenAIResponsesHistoryPayload("openai-codex", [ { type: "reasoning", encrypted_content: "enc_123" }, @@ -267,7 +267,7 @@ function makeAssistantMessage( items: Record[], incremental = false, provider: "openai" | "openai-codex" | "github-copilot" = "openai", - model = provider === "openai-codex" ? "gpt-5.2-codex" : provider === "github-copilot" ? "gpt-5.4" : "gpt-5-mini", + model = provider === "openai-codex" ? "gpt-5.5" : provider === "github-copilot" ? "gpt-5.4" : "gpt-5-mini", ) { return { role: "assistant" as const, @@ -428,7 +428,7 @@ describe("OpenAI responses history payload", () => { }); assertWireOrder(openaiItems); - const codexModel = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.2-codex"); + const codexModel = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.5"); const codexItems = convertCodexResponsesMessages(codexModel, makeContext("openai-codex")); assertWireOrder(codexItems); }); @@ -810,7 +810,7 @@ describe("OpenAI responses history payload", () => { }); it("prefers assistant native history snapshots for openai-codex-responses", async () => { - const model = getBundledModel("openai-codex", "gpt-5.2-codex") as Model<"openai-codex-responses">; + const model = getBundledModel("openai-codex", "gpt-5.5") as Model<"openai-codex-responses">; const payload = (await captureCodexPayload(model, codexAssistantSnapshotContext)) as { input?: unknown[] }; expect(payload.input).toEqual([ ...snapshotHistoryItems, @@ -1279,7 +1279,7 @@ describe("OpenAI responses history payload", () => { content: [{ type: "toolCall", id: callId, name: "read", arguments: { path: "README.md" } }], api: "openai-codex-responses", provider: "openai-codex", - model: "gpt-5.2-codex", + model: "gpt-5.5", usage: { input: 0, output: 0, @@ -1303,7 +1303,7 @@ describe("OpenAI responses history payload", () => { { role: "user", content: "Resume", timestamp: Date.now() }, ], }; - const model = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.2-codex"); + const model = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.5"); const payload = (await captureCodexPayload(model, context)) as { input?: unknown[] }; const functionCallItem = findResponsesInputItem(payload.input, "function_call"); const functionCallOutputItem = findResponsesInputItem(payload.input, "function_call_output"); diff --git a/packages/ai/test/openai-responses-openrouter.test.ts b/packages/ai/test/openai-responses-openrouter.test.ts index 71a9c7396..a57561c67 100644 --- a/packages/ai/test/openai-responses-openrouter.test.ts +++ b/packages/ai/test/openai-responses-openrouter.test.ts @@ -19,7 +19,14 @@ const context: Context = { messages: [{ role: "user", content: "ping", timestamp: 0 }], }; -function createSseResponse(): Response { +function createSseResponse( + usage: Record = { + input_tokens: 1, + output_tokens: 1, + total_tokens: 2, + input_tokens_details: { cached_tokens: 0 }, + }, +): Response { return new Response( `data: ${JSON.stringify({ type: "response.output_item.added", @@ -37,12 +44,7 @@ function createSseResponse(): Response { type: "response.completed", response: { status: "completed", - usage: { - input_tokens: 1, - output_tokens: 1, - total_tokens: 2, - input_tokens_details: { cached_tokens: 0 }, - }, + usage, }, })}\n\n`, { status: 200, headers: { "content-type": "text/event-stream" } }, @@ -297,6 +299,43 @@ describe("OpenRouter pseudo API dual-surface request parity", () => { }); describe("OpenRouter Responses request shape", () => { + it("uses OpenRouter's reported account charge instead of the catalog estimate", async () => { + const providerCost = 0.73; + const fetchMock: FetchImpl = vi.fn(async () => + createSseResponse({ + input_tokens: 1_000_000, + output_tokens: 100_000, + total_tokens: 1_100_000, + input_tokens_details: { cached_tokens: 0 }, + cost: providerCost, + }), + ); + const stream = streamOpenAIResponses( + buildOpenRouterResponsesModel({ + cost: { input: 0.435, output: 0.87, cacheRead: 0.003625, cacheWrite: 0 }, + }), + context, + { apiKey: "test-key", fetch: fetchMock }, + ); + let message: AssistantMessage | undefined; + for await (const event of stream) { + if (event.type === "done") { + message = event.message; + break; + } + if (event.type === "error") throw event.error; + } + if (!message) throw new Error("Expected completed OpenRouter response"); + + expect(message.usage.cost.total).toBe(providerCost); + const componentTotal = + message.usage.cost.input + + message.usage.cost.output + + message.usage.cost.cacheRead + + message.usage.cost.cacheWrite; + expect(componentTotal).toBeCloseTo(providerCost); + }); + it("appends openrouterVariant only when the resolved model id has no variant after the final slash", async () => { const suffixed = await captureRequest(buildOpenRouterResponsesModel(), { openrouterVariant: "nitro" }); expect(suffixed.body.model).toBe("anthropic/claude-haiku-latest:nitro"); diff --git a/packages/ai/test/openai-responses-parallel-tool-result-images.test.ts b/packages/ai/test/openai-responses-parallel-tool-result-images.test.ts new file mode 100644 index 000000000..f2119f568 --- /dev/null +++ b/packages/ai/test/openai-responses-parallel-tool-result-images.test.ts @@ -0,0 +1,141 @@ +import { describe, expect, it } from "bun:test"; +import { convertCodexResponsesMessages } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import type { ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire"; +import { buildResponsesInput } from "@oh-my-pi/pi-ai/providers/openai-shared"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const genericModel = buildModel({ + id: "moonshotai/kimi-k3", + name: "Kimi K3", + api: "openai-responses", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + reasoning: false, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_000, +}); + +const codexModel = buildModel({ + id: "gpt-5.5", + name: "Codex Test", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.com/backend-api/codex/responses", + reasoning: true, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 100_000, + compat: { supportsImageDetailOriginal: true }, +}); + +const zeroUsage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function makeContext(model: Model): Context { + return { + messages: [ + { + role: "assistant", + content: [ + { type: "toolCall", id: "call_read_36", name: "read", arguments: { path: "a.png" } }, + { type: "toolCall", id: "call_read_37", name: "read", arguments: { path: "b.png" } }, + { type: "toolCall", id: "call_bash_38", name: "bash", arguments: { command: "true" } }, + ], + api: model.api, + provider: model.provider, + model: model.id, + usage: zeroUsage, + stopReason: "toolUse", + timestamp: 1, + }, + { + role: "toolResult", + toolCallId: "call_read_36", + toolName: "read", + content: [ + { type: "text", text: "first" }, + { type: "image", mimeType: "image/png", data: "AAAA" }, + ], + isError: false, + timestamp: 2, + }, + { + role: "toolResult", + toolCallId: "call_read_37", + toolName: "read", + content: [ + { type: "text", text: "second" }, + { type: "image", mimeType: "image/png", data: "BBBB" }, + ], + isError: false, + timestamp: 3, + }, + { + role: "toolResult", + toolCallId: "call_bash_38", + toolName: "bash", + content: [{ type: "text", text: "done" }], + isError: false, + timestamp: 4, + }, + { role: "user", content: "continue", timestamp: 5 }, + ], + }; +} + +function expectOrderedToolResults(items: ResponseInput): void { + expect(items.slice(0, 3).map(item => ("call_id" in item ? item.call_id : undefined))).toEqual([ + "call_read_36", + "call_read_37", + "call_bash_38", + ]); + expect(items.slice(3)).toEqual([ + { type: "function_call_output", call_id: "call_read_36", output: "first" }, + { type: "function_call_output", call_id: "call_read_37", output: "second" }, + { type: "function_call_output", call_id: "call_bash_38", output: "done" }, + { + role: "user", + content: [ + { type: "input_text", text: "Attached image(s) from tool result:" }, + { type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" }, + ], + }, + { + role: "user", + content: [ + { type: "input_text", text: "Attached image(s) from tool result:" }, + { type: "input_image", detail: "auto", image_url: "data:image/png;base64,BBBB" }, + ], + }, + { role: "user", content: [{ type: "input_text", text: "continue" }] }, + ]); +} + +describe("parallel Responses tool-result images", () => { + it("keeps generic Responses outputs ahead of synthetic image messages", () => { + const items = buildResponsesInput({ + model: genericModel, + context: makeContext(genericModel), + strictResponsesPairing: true, + supportsImageDetailOriginal: true, + }); + + expectOrderedToolResults(items); + }); + + it("keeps Codex Responses outputs ahead of synthetic image messages", () => { + const items = convertCodexResponsesMessages(codexModel, makeContext(codexModel)); + + expectOrderedToolResults(items); + }); +}); diff --git a/packages/ai/test/openai-responses-sampling-params.test.ts b/packages/ai/test/openai-responses-sampling-params.test.ts new file mode 100644 index 000000000..cedab6f30 --- /dev/null +++ b/packages/ai/test/openai-responses-sampling-params.test.ts @@ -0,0 +1,70 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { streamSimple } from "@oh-my-pi/pi-ai/stream"; +import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; + +function mockSseFetch(): { fetchMock: FetchImpl; captured: Record } { + const captured: Record = {}; + const fetchMock: FetchImpl = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => { + const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record) : {}; + Object.assign(captured, body); + const event = { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 1, + output_tokens: 1, + total_tokens: 2, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }; + return new Response(`data: ${JSON.stringify(event)}\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); + }); + return { fetchMock, captured }; +} + +const ctx: Context = { + systemPrompt: ["hi"], + messages: [{ role: "user", content: "ping", timestamp: Date.now() }], +}; + +async function drain(model: Model<"openai-responses">): Promise> { + const { fetchMock, captured } = mockSseFetch(); + const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0 }); + for await (const event of stream) { + if (event.type === "done" || event.type === "error") break; + } + return captured; +} + +afterEach(() => { + vi.restoreAllMocks(); +}); + +describe("openai-responses sampling-param gating (#5606)", () => { + it("omits temperature for OpenAI reasoning models that reject it", async () => { + const model = getBundledModel("openai", "gpt-5") as Model<"openai-responses">; + expect(model.compat.supportsSamplingParams).toBe(false); + const body = await drain(model); + expect(body).not.toHaveProperty("temperature"); + }); + + it("omits temperature for GitHub Copilot gpt-5.6 (the reported model)", async () => { + const model = getBundledModel("github-copilot", "gpt-5.6-luna") as Model<"openai-responses">; + expect(model.compat.supportsSamplingParams).toBe(false); + const body = await drain(model); + expect(body).not.toHaveProperty("temperature"); + }); + + it("still forwards temperature for non-restricted OpenAI models", async () => { + const model = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">; + expect(model.compat.supportsSamplingParams).toBe(true); + const body = await drain(model); + expect(body.temperature).toBe(0); + }); +}); diff --git a/packages/ai/test/openai-responses-stateful.test.ts b/packages/ai/test/openai-responses-stateful.test.ts index 8f6fb1b1a..8871f0961 100644 --- a/packages/ai/test/openai-responses-stateful.test.ts +++ b/packages/ai/test/openai-responses-stateful.test.ts @@ -230,6 +230,58 @@ describe("openai-responses stateful chaining", () => { expect(JSON.stringify(sentRequests[2]?.input)).toContain("First question"); expect(JSON.stringify(sentRequests[2]?.input)).toContain("Second question"); }); + it("retries a blocked invalid_prompt previous_response_id with the full transcript", async () => { + const sentRequests: Array> = []; + const fetchMock = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + const request = JSON.parse(String(init?.body)) as Record; + sentRequests.push(request); + if (typeof request.previous_response_id === "string") { + return new Response( + JSON.stringify({ + error: { + message: "Request blocked.", + type: "invalid_request_error", + code: "invalid_prompt", + }, + }), + { status: 400, headers: { "content-type": "application/json" } }, + ); + } + return createStatefulSse(`Answer ${sentRequests.length}`, `resp_${sentRequests.length}`); + }) as FetchImpl; + const providerSessionState = new Map(); + const options = { + apiKey: "test-key", + sessionId: "stateful-blocked-session", + providerSessionState, + statefulResponses: true, + reasoning: "low" as const, + fetch: fetchMock, + }; + + const firstUser = { role: "user" as const, content: "First question", timestamp: 1000 }; + const firstResponse = await streamOpenAIResponses( + model, + { systemPrompt, messages: [firstUser] }, + options, + ).result(); + const secondResponse = await streamOpenAIResponses( + model, + { + systemPrompt, + messages: [firstUser, firstResponse, { role: "user", content: "Second question", timestamp: 1001 }], + }, + options, + ).result(); + + expect(secondResponse.stopReason).toBe("stop"); + expect(JSON.stringify(secondResponse.content)).toContain("Answer 3"); + expect(sentRequests).toHaveLength(3); + expect(sentRequests[1]?.previous_response_id).toBe("resp_1"); + expect(sentRequests[2]?.previous_response_id).toBeUndefined(); + expect(JSON.stringify(sentRequests[2]?.input)).toContain("First question"); + expect(JSON.stringify(sentRequests[2]?.input)).toContain("Second question"); + }); it("disables chaining for the session after repeated stale failures and stops forcing store", async () => { const sentRequests: Array> = []; diff --git a/packages/ai/test/openai-responses-stream-retry.test.ts b/packages/ai/test/openai-responses-stream-retry.test.ts new file mode 100644 index 000000000..b768b9363 --- /dev/null +++ b/packages/ai/test/openai-responses-stream-retry.test.ts @@ -0,0 +1,564 @@ +import { describe, expect, it, vi } from "bun:test"; +import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; +import type { + AssistantMessageEvent, + AssistantMessageEventStream, + Context, + FetchImpl, + Model, + ProviderSessionState, +} from "@oh-my-pi/pi-ai/types"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; + +const model = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">; +const firstUser = { role: "user" as const, content: "Read the file", timestamp: 1_000 }; +const context: Context = { messages: [firstUser] }; + +function createSseResponse(events: unknown[]): Response { + return new Response(`${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +function createTruncatedPendingToolResponse(): Response { + const prefix = [ + { type: "response.created", response: { id: "resp_partial", status: "in_progress" } }, + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_partial", + call_id: "call_partial", + name: "read", + arguments: "", + status: "in_progress", + }, + }, + ]; + const truncatedEvent = 'data: {"type":"response.function_call_arguments.delta","item_id":"fc_partial","delta":'; + return new Response(`${prefix.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n${truncatedEvent}`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +function createTruncatedReasoningPartDoneResponse(): Response { + const prefix = [ + { type: "response.created", response: { id: "resp_reasoning", status: "in_progress" } }, + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "reasoning_partial", summary: [], status: "in_progress" }, + }, + { + type: "response.reasoning_summary_part.added", + item_id: "reasoning_partial", + output_index: 0, + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_part.done", + item_id: "reasoning_partial", + output_index: 0, + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + ]; + const truncatedEvent = 'data: {"type":"response.output_text.delta","item_id":"missing","delta":'; + return new Response(`${prefix.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n${truncatedEvent}`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +function createCompletedToolResponse(responseId = "resp_retry"): Response { + const argumentsJson = JSON.stringify({ path: "README.md" }); + return createSseResponse([ + { type: "response.created", response: { id: responseId, status: "in_progress" } }, + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_recovered", + call_id: "call_recovered", + name: "read", + arguments: "", + status: "in_progress", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_recovered", + delta: argumentsJson, + }, + { + type: "response.function_call_arguments.done", + output_index: 0, + item_id: "fc_recovered", + arguments: argumentsJson, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "function_call", + id: "fc_recovered", + call_id: "call_recovered", + name: "read", + arguments: argumentsJson, + status: "completed", + }, + }, + { + type: "response.completed", + response: { + id: responseId, + status: "completed", + usage: { input_tokens: 5, output_tokens: 3, total_tokens: 8, input_tokens_details: { cached_tokens: 0 } }, + }, + }, + ]); +} + +function createCompletedTextResponse(text: string, responseId: string): Response { + return createSseResponse([ + { type: "response.created", response: { id: responseId, status: "in_progress" } }, + { + type: "response.output_item.added", + output_index: 0, + item: { type: "message", id: `msg_${responseId}`, role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.output_text.delta", output_index: 0, item_id: `msg_${responseId}`, delta: text }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "message", + id: `msg_${responseId}`, + role: "assistant", + status: "completed", + content: [{ type: "output_text", text }], + }, + }, + { type: "response.completed", response: { id: responseId, status: "completed" } }, + ]); +} + +function createGatedTextAndToolResponse(): { + response: Response; + terminalRequested: Promise; + releaseTerminal: () => void; +} { + const nonTerminalEvents = [ + { type: "response.created", response: { id: "resp_live", status: "in_progress" } }, + { + type: "response.output_item.added", + output_index: 0, + item: { type: "message", id: "msg_live", role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.output_text.delta", output_index: 0, item_id: "msg_live", delta: "draft" }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "message", + id: "msg_live", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "draft" }], + }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { + type: "function_call", + id: "fc_live", + call_id: "call_live", + name: "read", + arguments: "", + status: "in_progress", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 1, + item_id: "fc_live", + delta: '{"path":"README', + }, + ]; + const terminalEvents = [ + { + type: "response.function_call_arguments.done", + output_index: 1, + item_id: "fc_live", + arguments: '{"path":"README.md"}', + }, + { + type: "response.output_item.done", + output_index: 1, + item: { + type: "function_call", + id: "fc_live", + call_id: "call_live", + name: "read", + arguments: '{"path":"README.md"}', + status: "completed", + }, + }, + { type: "response.completed", response: { id: "resp_live", status: "completed" } }, + ]; + const terminalGate = Promise.withResolvers(); + const terminalRequest = Promise.withResolvers(); + let sentNonTerminal = false; + const body = new ReadableStream( + { + async pull(controller) { + if (!sentNonTerminal) { + sentNonTerminal = true; + controller.enqueue( + new TextEncoder().encode( + `${nonTerminalEvents.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`, + ), + ); + return; + } + terminalRequest.resolve(); + await terminalGate.promise; + controller.enqueue( + new TextEncoder().encode( + `${terminalEvents.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`, + ), + ); + controller.close(); + }, + }, + { highWaterMark: 0 }, + ); + return { + response: new Response(body, { status: 200, headers: { "content-type": "text/event-stream" } }), + terminalRequested: terminalRequest.promise, + releaseTerminal: terminalGate.resolve, + }; +} + +function parseBody(init: RequestInit | undefined): Record { + return JSON.parse(String(init?.body)) as Record; +} + +async function collectEvents(stream: AssistantMessageEventStream): Promise { + const events: AssistantMessageEvent[] = []; + for await (const event of stream) events.push(event); + return events; +} + +describe("OpenAI Responses transient stream retry", () => { + it("retries a truncated pending tool call with a fresh request and clean state", async () => { + const sentRequests: Array> = []; + let attempt = 0; + let payloadCalls = 0; + const fetchMock = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + sentRequests.push(parseBody(init)); + attempt++; + if (attempt === 1) return createTruncatedPendingToolResponse(); + if (attempt === 2) return createCompletedToolResponse(); + return createCompletedTextResponse("Follow-up", "resp_followup"); + }) as FetchImpl; + const providerSessionState = new Map(); + const options = { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + providerSessionState, + sessionId: "stream-retry-session", + statefulResponses: true, + onPayload: (payload: unknown) => { + payloadCalls++; + return { ...(payload as Record), metadata: { retry_test: "preserved" } }; + }, + }; + + const responseStream = streamOpenAIResponses(model, context, options); + const events = await collectEvents(responseStream); + const result = await responseStream.result(); + + expect(fetchMock).toHaveBeenCalledTimes(2); + expect(payloadCalls).toBe(1); + expect(sentRequests[0]?.metadata).toEqual({ retry_test: "preserved" }); + expect(sentRequests[1]).toEqual(sentRequests[0]); + expect(result.stopReason).toBe("toolUse"); + expect(JSON.parse(JSON.stringify(result.content))).toEqual([ + { type: "toolCall", id: "call_recovered|fc_recovered", name: "read", arguments: { path: "README.md" } }, + ]); + expect(events.map(event => event.type)).toEqual([ + "start", + "toolcall_start", + "toolcall_delta", + "toolcall_end", + "done", + ]); + expect(JSON.stringify(result.providerPayload)).not.toContain("partial"); + + const followup = await streamOpenAIResponses( + model, + { + messages: [firstUser, result, { role: "user", content: "What did it contain?", timestamp: 1_001 }], + }, + options, + ).result(); + expect(followup.stopReason).toBe("stop"); + expect(sentRequests[2]?.previous_response_id).toBe("resp_retry"); + }); + + it("falls back to full transcript when a fresh stream retry finds a stale chain baseline", async () => { + const sentRequests: Array> = []; + const fetchMock = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + const request = parseBody(init); + sentRequests.push(request); + switch (sentRequests.length) { + case 1: + return createCompletedTextResponse("Baseline", "resp_baseline"); + case 2: + return createTruncatedPendingToolResponse(); + case 3: + return new Response( + JSON.stringify({ + error: { + message: "Previous response with id 'resp_baseline' not found.", + type: "invalid_request_error", + param: "previous_response_id", + code: "previous_response_not_found", + }, + }), + { status: 404, headers: { "content-type": "application/json" } }, + ); + case 4: + return createCompletedTextResponse("Recovered", "resp_recovered"); + default: + return createCompletedTextResponse("Follow-up", "resp_followup"); + } + }) as FetchImpl; + const providerSessionState = new Map(); + const options = { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + providerSessionState, + sessionId: "stream-retry-stale-chain-session", + statefulResponses: true, + }; + + const baseline = await streamOpenAIResponses(model, context, options).result(); + const secondUser = { role: "user" as const, content: "Continue after baseline", timestamp: 1_001 }; + const responseStream = streamOpenAIResponses(model, { messages: [firstUser, baseline, secondUser] }, options); + const events = await collectEvents(responseStream); + const recovered = await responseStream.result(); + + expect(fetchMock).toHaveBeenCalledTimes(4); + expect(sentRequests[0]?.previous_response_id).toBeUndefined(); + expect(sentRequests[1]?.previous_response_id).toBe("resp_baseline"); + expect(sentRequests[2]).toEqual(sentRequests[1]); + expect(JSON.stringify(sentRequests[1]?.input)).toContain("Continue after baseline"); + expect(JSON.stringify(sentRequests[1]?.input)).not.toContain("Read the file"); + expect(sentRequests[3]?.previous_response_id).toBeUndefined(); + expect(sentRequests[3]?.store).toBe(true); + expect(JSON.stringify(sentRequests[3]?.input)).toContain("Read the file"); + expect(JSON.stringify(sentRequests[3]?.input)).toContain("Baseline"); + expect(JSON.stringify(sentRequests[3]?.input)).toContain("Continue after baseline"); + expect(recovered.responseId).toBe("resp_recovered"); + expect(events.map(event => event.type)).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); + + const followup = await streamOpenAIResponses( + model, + { + messages: [ + firstUser, + baseline, + secondUser, + recovered, + { role: "user", content: "One more question", timestamp: 1_002 }, + ], + }, + options, + ).result(); + expect(followup.stopReason).toBe("stop"); + expect(fetchMock).toHaveBeenCalledTimes(5); + expect(sentRequests[4]?.previous_response_id).toBe("resp_recovered"); + }); + + it("forwards text and tool deltas live with their delta-time partial state", async () => { + const gated = createGatedTextAndToolResponse(); + const fetchMock = vi.fn(async () => gated.response) as FetchImpl; + const responseStream = streamOpenAIResponses(model, context, { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + }); + const nonTerminalEvents = (async () => { + const observed: Array<{ type: AssistantMessageEvent["type"]; content: unknown }> = []; + for await (const event of responseStream) { + observed.push({ + type: event.type, + content: + event.type !== "start" && "partial" in event ? structuredClone(event.partial.content) : undefined, + }); + if (event.type === "toolcall_delta") return observed; + } + throw new Error("stream ended before the tool delta"); + })(); + + await gated.terminalRequested; + const observedBeforeTerminal = await nonTerminalEvents; + const deltaText = { type: "text", text: "draft", textSignature: JSON.stringify({ v: 1, id: "msg_live" }) }; + expect(observedBeforeTerminal).toEqual([ + { type: "start", content: undefined }, + { type: "text_start", content: [deltaText] }, + { type: "text_delta", content: [deltaText] }, + { type: "text_end", content: [deltaText] }, + { + type: "toolcall_start", + content: [deltaText, { type: "toolCall", id: "call_live|fc_live", name: "read", arguments: {} }], + }, + { + type: "toolcall_delta", + content: [ + deltaText, + { type: "toolCall", id: "call_live|fc_live", name: "read", arguments: { path: "README" } }, + ], + }, + ]); + + gated.releaseTerminal(); + const result = await responseStream.result(); + expect(result.stopReason).toBe("toolUse"); + expect(JSON.parse(JSON.stringify(result.content[1]))).toEqual({ + type: "toolCall", + id: "call_live|fc_live", + name: "read", + arguments: { path: "README.md" }, + }); + }); + + it("does not retry after a tool argument delta was emitted", async () => { + const partialWithDelta = createSseResponse([ + { type: "response.created", response: { id: "resp_partial", status: "in_progress" } }, + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_partial", + call_id: "call_partial", + name: "read", + arguments: "", + status: "in_progress", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_partial", + delta: '{"path":"README.md"}', + }, + ]); + const fetchMock = vi.fn(async () => partialWithDelta) as FetchImpl; + + const result = await streamOpenAIResponses(model, context, { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + }).result(); + + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(result.stopReason).toBe("error"); + }); + + it("does not retry after a reasoning summary part completion emitted thinking", async () => { + const fetchMock = vi.fn(async () => createTruncatedReasoningPartDoneResponse()) as FetchImpl; + const responseStream = streamOpenAIResponses(model, context, { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + }); + + const events = await collectEvents(responseStream); + const result = await responseStream.result(); + + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(events.map(event => event.type)).toEqual(["start", "thinking_start", "thinking_delta", "error"]); + expect(result.stopReason).toBe("error"); + }); + + it("bounds repeated pre-output stream corruption to one retry", async () => { + const fetchMock = vi.fn(async () => createTruncatedPendingToolResponse()) as FetchImpl; + + const result = await streamOpenAIResponses(model, context, { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + }).result(); + + expect(fetchMock).toHaveBeenCalledTimes(2); + expect(result.stopReason).toBe("error"); + }); + + it("honors caller abort during the retry wait", async () => { + const controller = new AbortController(); + const fetchMock = vi.fn(async () => createTruncatedPendingToolResponse()) as FetchImpl; + + const responseStream = streamOpenAIResponses(model, context, { + apiKey: "test-key", + fetch: fetchMock, + signal: controller.signal, + providerRetryWait: async () => controller.abort(), + }); + const events = await collectEvents(responseStream); + const result = await responseStream.result(); + + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(result.stopReason).toBe("aborted"); + expect(events.map(event => event.type)).toEqual(["start", "error"]); + expect(result.content).toEqual([]); + expect(result.responseId).toBeUndefined(); + expect(result.providerPayload).toBeUndefined(); + expect(result.usage).toEqual({ + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }); + expect(result.ttft).toBeUndefined(); + }); + + for (const error of [ + { code: "invalid_request_error", message: "Tool schema is invalid" }, + { code: "insufficient_quota", message: "Persistent quota exhausted" }, + ]) { + it(`does not retry terminal ${error.code} failures`, async () => { + const fetchMock = vi.fn(async () => + createSseResponse([ + { + type: "response.failed", + response: { id: "resp_failed", status: "failed", error }, + }, + ]), + ) as FetchImpl; + + const result = await streamOpenAIResponses(model, context, { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + }).result(); + + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(result.stopReason).toBe("error"); + }); + } +}); diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index 4272195a6..8b9f2e4b4 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -1,9 +1,9 @@ // Terminal-event contracts for `processResponsesStream`: // // 1. `response.incomplete` is a terminal frame (max_output_tokens / content -// filter truncation). It must populate usage and map to stopReason -// "length" — previously it was ignored entirely, so truncated responses -// reported stopReason "stop" with zero usage and no cost. +// filter truncation). It must populate usage and normally map to stopReason +// "length", while a fully streamed function call remains executable as +// "toolUse". // 2. `response.output_item.done` for a custom_tool_call must persist the final // input on the stored content block and drop the transient `partialJson` // accumulation buffer, mirroring the function_call branch. @@ -120,6 +120,325 @@ describe("processResponsesStream: terminal events", () => { expect(output.content).toEqual([expect.objectContaining({ type: "text", text: "Hello, trunc" })]); }); + test("promotes max-output incomplete function calls with strict-complete arguments", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_complete", + call_id: "call_complete", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_complete", + delta: '{"path":"complete.txt"} \n\t', + }, + { + type: "response.incomplete", + response: { + id: "resp_complete_call", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("toolUse"); + expect(output.content).toHaveLength(1); + const block = output.content[0]; + if (block?.type !== "toolCall") throw new Error("expected a toolCall block"); + expect(block.arguments).toEqual({ path: "complete.txt" }); + }); + + test("keeps max-output incomplete function calls at length when only output_item.done closes arguments", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_closed", + call_id: "call_closed", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_closed", + delta: '{"path":"closed.txt"}', + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "function_call", + id: "fc_closed", + call_id: "call_closed", + name: "read", + arguments: '{"path":"closed.txt"}', + }, + }, + { + type: "response.incomplete", + response: { + id: "resp_closed_call", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + expect(output.content).toEqual([ + expect.objectContaining({ type: "toolCall", arguments: { path: "closed.txt" } }), + ]); + }); + + test("promotes max-output incomplete custom tool calls closed by input done", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + const patch = "*** Begin Patch\n*** End Patch"; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "custom_tool_call", + id: "ctc_closed", + call_id: "call_custom_closed", + name: "apply_patch", + input: "", + }, + }, + { + type: "response.custom_tool_call_input.delta", + output_index: 0, + item_id: "ctc_closed", + delta: patch, + }, + { + type: "response.custom_tool_call_input.done", + output_index: 0, + item_id: "ctc_closed", + input: patch, + }, + { + type: "response.incomplete", + response: { + id: "resp_closed_custom", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("toolUse"); + expect(output.content).toEqual([ + expect.objectContaining({ type: "toolCall", customWireName: "apply_patch", arguments: { input: patch } }), + ]); + }); + + test("keeps max-output incomplete custom tools at length when only output_item.done closes input", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + const patch = "*** Begin Patch\n*** End Patch"; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "custom_tool_call", + id: "ctc_output_done", + call_id: "call_custom_output_done", + name: "apply_patch", + input: "", + }, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "custom_tool_call", + id: "ctc_output_done", + call_id: "call_custom_output_done", + name: "apply_patch", + input: patch, + }, + }, + { + type: "response.incomplete", + response: { + id: "resp_output_done_custom", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + expect(output.content).toEqual([ + expect.objectContaining({ type: "toolCall", customWireName: "apply_patch", arguments: { input: patch } }), + ]); + }); + + test("keeps max-output incomplete turns at length when any function call has a JSON prefix", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_first", + call_id: "call_first", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_first", + delta: '{"path":"complete.txt"}', + }, + { + type: "response.output_item.added", + output_index: 1, + item: { + type: "function_call", + id: "fc_second", + call_id: "call_second", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 1, + item_id: "fc_second", + delta: '{"path":"truncated.txt"', + }, + { + type: "response.incomplete", + response: { + id: "resp_truncated_call", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + }); + + test("keeps max-output incomplete unfinished custom-tool input at length", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { + type: "function_call", + id: "fc_with_custom", + call_id: "call_with_custom", + name: "read", + arguments: "", + }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_with_custom", + delta: '{"path":"complete.txt"}', + }, + { + type: "response.output_item.added", + output_index: 1, + item: { + type: "custom_tool_call", + id: "ctc_unfinished", + call_id: "call_unfinished", + name: "apply_patch", + input: "", + }, + }, + { + type: "response.custom_tool_call_input.delta", + output_index: 1, + item_id: "ctc_unfinished", + delta: "*** Begin Patch", + }, + { + type: "response.incomplete", + response: { + id: "resp_unfinished_custom", + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.stopReason).toBe("length"); + const customCall = output.content.find(block => block.type === "toolCall" && block.customWireName !== undefined); + expect(customCall).toEqual( + expect.objectContaining({ + type: "toolCall", + customWireName: "apply_patch", + arguments: { input: "*** Begin Patch" }, + }), + ); + }); + for (const testCase of [ { name: "absent terminal content preserves streamed text", @@ -400,6 +719,41 @@ describe("processResponsesStream: lost output_item.added recovery", () => { expect(end?.content).toBe("Recovered text"); }); + test("normalizes a completed native image generation call into visible assistant content", async () => { + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (event: unknown) => emitted.push(event as EmittedEvent), end: () => {} } as never; + const data = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII="; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "image_generation_call", + id: "ig_1", + status: "completed", + result: data, + }, + }, + { type: "response.completed", response: { id: "resp_image", status: "completed" } }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.content).toEqual([{ type: "image", data, mimeType: "image/png" }]); + const end = emitted.find(event => event.type === "image_end"); + expect(end).toEqual({ + type: "image_end", + contentIndex: 0, + content: { type: "image", data, mimeType: "image/png" }, + partial: output, + }); + }); + test("routes reasoning finalization by output_index when item ids are absent", async () => { const output = makeOutput(); const stream = { push: () => {}, end: () => {} } as never; diff --git a/packages/ai/test/openai-responses-system-prompt.test.ts b/packages/ai/test/openai-responses-system-prompt.test.ts index f09888189..26103a45c 100644 --- a/packages/ai/test/openai-responses-system-prompt.test.ts +++ b/packages/ai/test/openai-responses-system-prompt.test.ts @@ -86,6 +86,16 @@ describe("openai-responses system prompt routing", () => { expect(input.every(m => m.role !== "system")).toBe(true); }); + it("redacts sensitive credentials in instructions", async () => { + const context: Context = { + systemPrompt: ["Token: gho_************************************"], + messages: [{ role: "user", content: "hi", timestamp: Date.now() }], + }; + const body = await captureRequestBody(gpt4oMiniModel, context); + + expect(body.instructions).toBe("Token: [github_token_redacted]"); + }); + it("omits instructions field when there is no system prompt", async () => { const context: Context = { systemPrompt: undefined, diff --git a/packages/ai/test/pi-native-client.test.ts b/packages/ai/test/pi-native-client.test.ts index 930c23166..194c02524 100644 --- a/packages/ai/test/pi-native-client.test.ts +++ b/packages/ai/test/pi-native-client.test.ts @@ -48,30 +48,39 @@ function stalledBody(bytes: Uint8Array[] = []): ReadableStream { } function delayedBody(chunks: Array<{ atMs: number; bytes: Uint8Array }>): ReadableStream { - let active = true; + let closed = false; + const timers: Timer[] = []; + const clearTimers = () => { + closed = true; + for (const timer of timers) clearTimeout(timer); + timers.length = 0; + }; return new ReadableStream({ start(controller) { + const enqueue = (bytes: Uint8Array) => { + if (!closed) controller.enqueue(bytes); + }; for (const chunk of chunks) { - setTimeout(() => { - if (!active) return; - try { - controller.enqueue(chunk.bytes); - } catch {} - }, chunk.atMs); + if (chunk.atMs <= 0) { + enqueue(chunk.bytes); + } else { + timers.push(setTimeout(() => enqueue(chunk.bytes), chunk.atMs)); + } } - setTimeout( - () => { - if (!active) return; - active = false; - try { - controller.close(); - } catch {} - }, - Math.max(...chunks.map(chunk => chunk.atMs)) + 1, + timers.push( + setTimeout( + () => { + if (!closed) { + clearTimers(); + controller.close(); + } + }, + Math.max(...chunks.map(chunk => chunk.atMs)) + 1, + ), ); }, cancel() { - active = false; + clearTimers(); }, }); } diff --git a/packages/ai/test/proxy.test.ts b/packages/ai/test/proxy.test.ts index 6ea0a78a9..0ee8d9946 100644 --- a/packages/ai/test/proxy.test.ts +++ b/packages/ai/test/proxy.test.ts @@ -63,6 +63,19 @@ async function waitForSocketClose(socket: net.Socket): Promise { const isProxyEnvKey = (k: string): boolean => k.startsWith("PI_PROXY") || k === "NO_PROXY" || k === "no_proxy"; +// NO_PROXY/no_proxy set at runtime are readable but hidden from Bun.env +// enumeration (Bun's fetch proxy layer intercepts them), so the sweep must +// name them explicitly instead of relying on for..in. +const HIDDEN_PROXY_KEYS = ["NO_PROXY", "no_proxy"]; + +function proxyEnvKeys(): Set { + const keys = new Set(HIDDEN_PROXY_KEYS); + for (const key in Bun.env) { + if (isProxyEnvKey(key)) keys.add(key); + } + return keys; +} + // Snapshot + clear every proxy-related env var so each test starts clean and // leaves nothing behind for later files. Provider-specific tests use unique // provider ids so the module-level resolver cache can never cross-contaminate. @@ -70,19 +83,14 @@ let saved: Record; beforeEach(() => { saved = {}; - for (const key in Bun.env) { - if (!isProxyEnvKey(key)) continue; + for (const key of proxyEnvKeys()) { saved[key] = Bun.env[key]; delete Bun.env[key]; } }); afterEach(() => { - const toDelete: string[] = []; - for (const key in Bun.env) { - if (isProxyEnvKey(key)) toDelete.push(key); - } - for (const key of toDelete) delete Bun.env[key]; + for (const key of proxyEnvKeys()) delete Bun.env[key]; for (const key in saved) { const value = saved[key]; if (value !== undefined) Bun.env[key] = value; @@ -190,6 +198,11 @@ describe("shouldBypassProxy NO_PROXY rules", () => { expect(shouldBypassProxy(new URL("https://api.sakana.ai/v1"))).toBe(false); expect(shouldBypassProxy(new URL("http://api.sakana.ai:8080/v1"))).toBe(true); }); + + it("uses port 443 for secure websocket targets", () => { + Bun.env.NO_PROXY = "api.sakana.ai:443"; + expect(shouldBypassProxy(new URL("wss://api.sakana.ai/v1"))).toBe(true); + }); }); describe("wrapFetchForProxy", () => { diff --git a/packages/ai/test/rate-limit-utils.test.ts b/packages/ai/test/rate-limit-utils.test.ts index 0b39b333b..51aad5121 100644 --- a/packages/ai/test/rate-limit-utils.test.ts +++ b/packages/ai/test/rate-limit-utils.test.ts @@ -61,6 +61,14 @@ describe("parseRateLimitReason", () => { ).toBe("QUOTA_EXHAUSTED"); }); + it("classifies Anthropic monthly spend limits as QUOTA_EXHAUSTED", () => { + expect( + parseRateLimitReason( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s monthly spend limit. Please try again later."}}', + ), + ).toBe("QUOTA_EXHAUSTED"); + }); + it("classifies OpenCode Go insufficient balance as QUOTA_EXHAUSTED", () => { expect( parseRateLimitReason("401 Insufficient balance. Manage your billing here: https://opencode.ai/workspace/demo"), @@ -121,6 +129,19 @@ describe("isUsageLimit", () => { ).toBe(true); }); + // Anthropic returns a `rate_limit_error` when the account's monthly spend + // cap is hit ("This request would exceed your account's monthly spend + // limit."). Without the `spend limit` branch the message classifies as a + // transient rate limit, so `isProviderRetryableError` retries it until the + // local deadline instead of surfacing the quota error (issue #4787). + it("detects Anthropic monthly spend-limit as a credential-rotatable usage limit", () => { + expect( + isUsageLimit( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s monthly spend limit. Please try again later."}}', + ), + ).toBe(true); + }); + it("detects bare 'quota reached' phrasing", () => { expect(isUsageLimit("quota reached")).toBe(true); expect(isUsageLimit("quota_reached")).toBe(true); @@ -213,6 +234,21 @@ describe("isUsageLimitOutcome", () => { expect(isUsageLimitOutcome(429, message)).toBe(true); }); + it("rotates on xAI Grok Build 402 usage-balance exhaustion regardless of status", () => { + const message = "402 Grok Build usage balance exhausted"; + expect(isUsageLimitOutcome(402, message)).toBe(true); + expect(isUsageLimitOutcome(undefined, message)).toBe(true); + expect(isUsageLimit(message)).toBe(true); + }); + + it("treats 402 as a usage-limit status (opaque body rotates, informative non-quota body does not)", () => { + expect(isUsageLimitStatus(402)).toBe(true); + expect(isUsageLimitOutcome(402, undefined)).toBe(true); + expect(isUsageLimitOutcome(402, "HTTP 402")).toBe(true); + expect(isUsageLimitOutcome(402, "A subscription is required for this endpoint")).toBe(false); + expect(isUsageLimit(new ProviderHttpError("HTTP 402", 402))).toBe(true); + }); + it("does not rotate on auth/invalid-request statuses with unrelated bodies", () => { expect(isUsageLimitOutcome(401, "Invalid API key")).toBe(false); expect(isUsageLimitOutcome(400, "invalid_request_error: model unsupported")).toBe(false); diff --git a/packages/ai/test/remote-auth-store.test.ts b/packages/ai/test/remote-auth-store.test.ts index 10be65fe6..9192b5209 100644 --- a/packages/ai/test/remote-auth-store.test.ts +++ b/packages/ai/test/remote-auth-store.test.ts @@ -150,6 +150,48 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => { remoteStore.close(); }); + test("invalidated OAuth tokens disable the remote row and rotate to a sibling", async () => { + serverStore!.upsertAuthCredentialForProvider("anthropic", { + type: "oauth", + access: "server-access-2", + refresh: "server-refresh-2", + expires: Date.now() + 120_000, + accountId: "account-2", + email: "b@example.com", + }); + await serverStorage!.reload(); + const seededRows = serverStore!.listAuthCredentials("anthropic"); + expect(seededRows).toHaveLength(2); + const failedRow = seededRows[0]; + if (failedRow?.credential.type !== "oauth") throw new Error("expected failed OAuth row"); + + const brokerClient = new AuthBrokerClient({ url: handle!.url, token }); + const initialResult = await brokerClient.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("expected snapshot"); + const remoteStore = new RemoteAuthCredentialStore({ + client: brokerClient, + initialSnapshot: initialResult.snapshot, + }); + const clientStorage = new AuthStorage(remoteStore); + const first = { + accessToken: failedRow.credential.access, + credentialId: failedRow.id, + }; + + const rotated = await clientStorage.rotateSessionCredential("anthropic", "invalidated-session", { + error: new Error("Encountered invalidated oauth token for user, failing request"), + apiKey: first.accessToken, + credentialId: first.credentialId, + }); + + expect(rotated).toBe(true); + expect(serverStore!.listAuthCredentials("anthropic").map(row => row.id)).not.toContain(first.credentialId); + const next = await clientStorage.getOAuthAccess("anthropic", "invalidated-session"); + expect(next?.credentialId).not.toBe(first.credentialId); + clientStorage.close(); + remoteStore.close(); + }); + test("RemoteAuthCredentialStore rejects writes from the client", () => { const remoteStore = new RemoteAuthCredentialStore({ client: new AuthBrokerClient({ url: handle!.url, token }), @@ -991,6 +1033,18 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => { expect(clientStorage.get("kagi")).toEqual({ type: "api_key", key: "new-key" }); clientStorage.close(); }); + test("snapshot with a login-sourced api_key passes client wire validation", async () => { + // Regression: keys stored via the /login flow carry `source: "login"`. + // exportSnapshot() forwards them verbatim; the client wire schema used + // to reject the field ("credentials[0].credential.source must be removed"). + await serverStorage!.set("custom-host", { type: "api_key", key: "sk-custom", source: "login" }); + + const brokerClient = new AuthBrokerClient({ url: handle!.url, token }); + const result = await brokerClient.fetchSnapshot(); + if (result.status !== 200) throw new Error("expected snapshot"); + const entry = result.snapshot.credentials.find(candidate => candidate.provider === "custom-host"); + expect(entry?.credential).toEqual({ type: "api_key", key: "sk-custom", source: "login" }); + }); test("client AuthStorage.remove disables every broker-side credential for the provider (logout)", async () => { serverStore!.saveApiKey("kagi", "k1"); diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index d3b031666..f9fa3ddc1 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -252,7 +252,7 @@ describe("normalizeSchemaForGoogle", () => { expect(sanitized.enum).toEqual([null]); }); - it("preserves a property schema literally named additionalProperties inside properties", () => { + it("coerces a boolean subschema literally named additionalProperties inside properties", () => { const sanitized = normalizeSchemaForGoogle({ type: "object", properties: { @@ -262,20 +262,81 @@ describe("normalizeSchemaForGoogle", () => { }) as Record; const properties = sanitized.properties as Record; + // The key survives (it is a property, not the stripped keyword), but its + // boolean subschema value coerces to the object form (issue #5604). expect(Object.hasOwn(properties, "additionalProperties")).toBe(true); - expect(properties.additionalProperties).toBe(false); + expect(properties.additionalProperties).toEqual({ not: {} }); }); - it("preserves boolean schemas for a single property literally named additionalProperties", () => { - const schema = { + it("coerces a boolean subschema for a single property literally named additionalProperties", () => { + const sanitized = normalizeSchemaForGoogle({ type: "object", properties: { additionalProperties: false, }, required: ["additionalProperties"], - } as const; + }) as Record; - expect(normalizeSchemaForGoogle(schema)).toEqual(schema); + const properties = sanitized.properties as Record; + expect(properties.additionalProperties).toEqual({ not: {} }); + expect(sanitized.required).toEqual(["additionalProperties"]); + }); + + it("coerces boolean subschemas to object equivalents on the Google wire (issue #5604)", () => { + const google = normalizeSchemaForGoogle({ + type: "object", + properties: { + propertyValue: true, + attributeValue: false, + }, + }) as Record; + + expect(google.properties).toEqual({ propertyValue: {}, attributeValue: { not: {} } }); + // Root-level and array-branch booleans are covered by the same choke point. + expect(normalizeSchemaForGoogle(true)).toEqual({}); + expect(normalizeSchemaForGoogle(false)).toEqual({ not: {} }); + expect(normalizeSchemaForGoogle({ anyOf: [true, { type: "string" }] })).toEqual({ + anyOf: [{}, { type: "string" }], + }); + }); + + it("strips draft-2019 conditional keywords the OpenAPI-style wire cannot model", () => { + // `dependentSchemas`/`dependencies`/`dependentRequired` have no Google + // OpenAPI Schema representation and are not caught by residual checks, so + // they must be dropped before serialization on both transports. + const input = { + type: "object", + properties: { propertyValue: true }, + dependentSchemas: { hasFoo: true }, + dependentRequired: { hasFoo: ["propertyValue"] }, + }; + const expected = { type: "object", properties: { propertyValue: {} } }; + + expect(normalizeSchemaForGoogle(input)).toEqual(expected); + expect(normalizeSchemaForCCA(input)).toEqual(expected); + + // MCP accepts native JSON Schema booleans, so it preserves them. + expect( + normalizeSchemaForMCP({ + type: "object", + dependentSchemas: { hasFoo: true, hasBar: false }, + }), + ).toEqual({ + type: "object", + dependentSchemas: { hasFoo: true, hasBar: false }, + }); + }); + + it("falls back when a false subschema produces unsupported `not` on the CCA wire", () => { + const fallback = { type: "object", properties: {} }; + + expect(normalizeSchemaForCCA(false)).toEqual(fallback); + expect(normalizeSchemaForCCA({ type: "object", properties: { attributeValue: false } })).toEqual(fallback); + // A property named `not` is a schema-map entry, not the unsupported keyword. + expect(normalizeSchemaForCCA({ type: "object", properties: { not: { type: "string" } } })).toEqual({ + type: "object", + properties: { not: { type: "string" } }, + }); }); it("inlines local $ref / $defs entries for Google compatibility", () => { @@ -1132,6 +1193,7 @@ function assertMfjsValid(node: unknown, path = "$"): void { for (const [i, entry] of node.entries()) assertMfjsValid(entry, `${path}[${i}]`); return; } + if (typeof node === "boolean") throw new Error(`MFJS requires an object schema at ${path}`); if (typeof node !== "object" || node === null) return; const obj = node as Record; for (const key of Object.keys(obj)) { @@ -1215,6 +1277,22 @@ describe("normalizeSchemaForMoonshot", () => { expect(props.limit).toEqual({ type: "integer", default: 10 }); }); + it("coerces boolean subschemas to MFJS object forms without changing boolean keywords", () => { + expect( + normalizeSchemaForMoonshot({ + type: "object", + properties: { allowed: true, forbidden: false }, + additionalProperties: false, + }), + ).toEqual({ + type: "object", + properties: { allowed: {}, forbidden: {} }, + additionalProperties: false, + }); + expect(normalizeSchemaForMoonshot(true)).toEqual({}); + expect(normalizeSchemaForMoonshot(false)).toEqual({}); + }); + it("folds oneOf into anyOf (the only MFJS combinator)", () => { const normalized = normalizeSchemaForMoonshot({ oneOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index 1ede17ae6..83e06c66d 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -199,6 +199,41 @@ describe("streamSimple resolver auth retry", () => { expect(keys).toEqual(["old-key", "new-key"]); }); + it("retries when Codex reports an invalidated OAuth token without an HTTP status", async () => { + const keys: unknown[] = []; + registerCustomApi( + API, + (_model: Model, _context: Context, options?: SimpleStreamOptions) => { + pushKey(keys, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + if (keys.length === 1) { + stream.push({ type: "start", partial: assistant() }); + stream.push({ + type: "error", + reason: "error", + error: assistantError("Encountered invalidated oauth token for user, failing request"), + }); + return; + } + ok(stream); + }); + return stream; + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { + apiKey: async ctx => (ctx.error === undefined ? "invalidated-key" : "healthy-key"), + }); + for await (const _event of stream) { + // drain + } + + expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); + expect(keys).toEqual(["invalidated-key", "healthy-key"]); + }); + it("does not retry after replay-unsafe content has been emitted", async () => { let retryResolves = 0; const failure = authError(); diff --git a/packages/ai/test/stream-markup-healing.test.ts b/packages/ai/test/stream-markup-healing.test.ts index 3401e058b..f9e59f421 100644 --- a/packages/ai/test/stream-markup-healing.test.ts +++ b/packages/ai/test/stream-markup-healing.test.ts @@ -523,6 +523,57 @@ describe("StreamMarkupHealing thinking pattern", () => { expect(heal("see

content
end")).toEqual({ text: "see
content
end", thinking: "" }); }); + // Issue #5665: a literal reasoning tag inside a Markdown inline-code span was + // read as a leaked boundary, splitting the visible row into + // text + thinking and corrupting the rendered Markdown. + it("keeps a literal think tag inside inline code as visible text", () => { + const literal = `<${"think"}>`; + const row = `| [#1203 MiniMax CN leaks \`${literal}\` text](https://x) | Fixed | PR merged |`; + expect(heal(row)).toEqual({ text: row, thinking: "" }); + }); + + it("keeps a literal think tag inside inline code when streamed char by char", () => { + const literal = `<${"think"}>`; + const row = `prefix \`${literal}\` suffix`; + expect(heal(...row)).toEqual({ text: row, thinking: "" }); + }); + + it("keeps a literal think tag inside a fenced code block as visible text", () => { + const literal = `<${"think"}>`; + const block = `\`\`\`md\n${literal}\n\`\`\`\nafter`; + expect(heal(block)).toEqual({ text: block, thinking: "" }); + }); + + // Issue #5665 (review follow-up): a fenced block only closes on its own fence + // line. An inline backtick run inside the block (a `` ``` `` string literal) + // must not exit code mode early and let a later literal think tag be healed. + it("keeps a fenced block open across an inner triple-backtick literal", () => { + const literal = `<${"think"}>literal`; + const block = `\`\`\`md\nconst fence = '\`\`\`';\n${literal}\n\`\`\`\nafter`; + expect(heal(block)).toEqual({ text: block, thinking: "" }); + expect(heal(...block)).toEqual({ text: block, thinking: "" }); + }); + + // Issue #5665 (review follow-up): CommonMark treats a fence indented by up to + // three spaces as fenced code. The scanner must still open a fenced block (not + // an inline span) so an inner triple-backtick literal does not close it early. + it("recognizes a fence indented up to three spaces as a fenced block", () => { + const literal = `<${"think"}>literal`; + for (const indent of ["", " ", " "]) { + const block = `${indent}\`\`\`md\nconst fence = '\`\`\`';\n${literal}\n${indent}\`\`\`\nafter`; + expect(heal(block)).toEqual({ text: block, thinking: "" }); + expect(heal(...block)).toEqual({ text: block, thinking: "" }); + } + }); + + it("still heals a leaked think tag outside inline code", () => { + const literal = `<${"think"}>`; + expect(heal(`before \`code\` ${literal}secret after`)).toEqual({ + text: "before `code` after", + thinking: "secret", + }); + }); + it("emits one balanced thinking boundary for a healed fence", () => { const scanner = new ThinkingInbandScanner(); const events: InbandScanEvent[] = [...scanner.feed("a```thinking\nx\n```b"), ...scanner.flush()]; diff --git a/packages/ai/test/synthetic-usage.test.ts b/packages/ai/test/synthetic-usage.test.ts new file mode 100644 index 000000000..10de2dcd5 --- /dev/null +++ b/packages/ai/test/synthetic-usage.test.ts @@ -0,0 +1,194 @@ +import { describe, expect, it } from "bun:test"; + +import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; +import type { UsageFetchContext, UsageFetchParams } from "@oh-my-pi/pi-ai/usage"; +import { syntheticUsageProvider } from "@oh-my-pi/pi-ai/usage/synthetic"; + +const FULL_FIXTURE = { + subscription: { limit: 500, requests: 12, renewsAt: "2026-07-10T11:33:46.399Z" }, + search: { hourly: { limit: 250, requests: 0, renewsAt: "2026-07-10T07:33:46.399Z" } }, + freeToolCalls: { limit: 0, requests: 0, renewsAt: "2026-07-11T06:33:46.405Z" }, + weeklyTokenLimit: { + nextRegenAt: "2026-07-10T08:17:04.000Z", + percentRemaining: 7.615, + maxCredits: "$24.00", + remainingCredits: "$1.82", + nextRegenCredits: "$0.48", + }, + rollingFiveHourLimit: { + nextTickAt: "2026-07-10T06:46:05.000Z", + tickPercent: 0.05, + remaining: 500, + max: 500, + limited: false, + }, +}; + +function makeCredential(): UsageFetchParams["credential"] { + return { + type: "api_key", + apiKey: "synthetic-test-key", + }; +} + +function makeCtx(payload: unknown, status = 200): UsageFetchContext { + const fetch: FetchImpl = async () => { + return new Response(JSON.stringify(payload), { + status, + headers: { "content-type": "application/json" }, + }); + }; + return { fetch }; +} + +function makeCtxThrow(): UsageFetchContext { + const fetch: FetchImpl = async () => { + throw new Error("Network error"); + }; + return { fetch }; +} + +describe("synthetic usage provider", () => { + it("happy path: full fixture returns exactly 2 limits — 5h rolling and weekly credits", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx(FULL_FIXTURE), + ); + + expect(report).not.toBeNull(); + expect(report!.limits).toHaveLength(2); + expect(report!.limits.map(l => l.id)).toEqual(["synthetic:requests:5h", "synthetic:usd:7d"]); + expect(report!.limits.map(l => l.label)).toEqual(["Synthetic Requests", "Synthetic Credits"]); + expect(report!.limits.map(l => l.scope.windowId)).toEqual(["5h", "7d"]); + expect(report!.limits.map(l => l.window?.durationMs)).toEqual([5 * 60 * 60 * 1000, 7 * 24 * 60 * 60 * 1000]); + }); + + it("5h rolling limit: used/remaining math (500 max, 500 remaining → 0 used, status ok)", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx(FULL_FIXTURE), + ); + + const fiveHour = report!.limits.find(l => l.id === "synthetic:requests:5h")!; + expect(fiveHour.amount.remaining).toBe(500); + expect(fiveHour.amount.limit).toBe(500); + expect(fiveHour.amount.used).toBe(0); + expect(fiveHour.amount.usedFraction).toBe(0); + expect(fiveHour.status).toBe("ok"); + }); + + it("5h rolling limit: regen rate folded into window label, tick time drives resetsAt", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx(FULL_FIXTURE), + ); + + const fiveHour = report!.limits.find(l => l.id === "synthetic:requests:5h")!; + // Single-line rendering: no notes; tickPercent 0.05 → "5%" in the window label. + expect(fiveHour.notes).toBeUndefined(); + expect(fiveHour.window?.label).toBe("5h · regen 5%/tick"); + // nextTickAt drives the countdown, rendered as "tick in …" (not a window reset). + expect(fiveHour.window?.resetsAt).toBe(Date.parse("2026-07-10T06:46:05.000Z")); + expect(fiveHour.window?.resetLabel).toBe("tick"); + }); + + it("weekly USD limit: parses dollar amounts, uses percentRemaining for usedFraction", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx(FULL_FIXTURE), + ); + + const weekly = report!.limits.find(l => l.id === "synthetic:usd:7d")!; + expect(weekly.amount.limit).toBeCloseTo(24.0); + expect(weekly.amount.remaining).toBeCloseTo(1.82); + // usedFraction = 1 - 7.615/100 ≈ 0.92385 + expect(weekly.amount.usedFraction).toBeCloseTo(1 - 7.615 / 100, 5); + expect(weekly.amount.unit).toBe("usd"); + // usedFraction ~0.924 ≥ 0.9 → "warning" + expect(weekly.status).toBe("warning"); + // Weekly credits regenerate incrementally too: "regen in …" + $/tick in the label. + expect(weekly.window?.resetLabel).toBe("regen"); + expect(weekly.window?.label).toBe("7d · regen $0.48/tick"); + expect(weekly.window?.resetsAt).toBe(Date.parse("2026-07-10T08:17:04.000Z")); + }); + + it("docs-minimal response (subscription-only) → null (no surfaceable limits)", async () => { + const minimal = { + subscription: { limit: 100, requests: 5, renewsAt: "2026-08-01T00:00:00.000Z" }, + }; + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx(minimal), + ); + // subscription is dropped; no 5h or weekly data → nothing to show + expect(report).toBeNull(); + }); + + it("limited: true on rollingFiveHourLimit → status is exhausted", async () => { + const payload = { + ...FULL_FIXTURE, + rollingFiveHourLimit: { ...FULL_FIXTURE.rollingFiveHourLimit, remaining: 0, limited: true }, + }; + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx(payload), + ); + + const fiveHour = report!.limits.find(l => l.id === "synthetic:requests:5h")!; + expect(fiveHour.status).toBe("exhausted"); + }); + + it("non-synthetic provider → returns null", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "openai", credential: makeCredential(), signal: undefined }, + makeCtx(FULL_FIXTURE), + ); + expect(report).toBeNull(); + }); + + it("non-api_key credential → returns null", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: { type: "oauth" } as UsageFetchParams["credential"], signal: undefined }, + makeCtx(FULL_FIXTURE), + ); + expect(report).toBeNull(); + }); + + it("HTTP error response → returns null", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx({ message: "Unauthorized" }, 401), + ); + expect(report).toBeNull(); + }); + + it("network error / thrown fetch → returns null", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtxThrow(), + ); + expect(report).toBeNull(); + }); + + it("malformed JSON payload → returns null", async () => { + const fetch: FetchImpl = async () => { + return new Response("not-json{{{{", { + status: 200, + headers: { "content-type": "application/json" }, + }); + }; + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + { fetch }, + ); + expect(report).toBeNull(); + }); + + it("empty object payload → returns null", async () => { + const report = await syntheticUsageProvider.fetchUsage!( + { provider: "synthetic", credential: makeCredential(), signal: undefined }, + makeCtx({}), + ); + expect(report).toBeNull(); + }); +}); diff --git a/packages/ai/test/transform-messages-malformed-tool-calls.test.ts b/packages/ai/test/transform-messages-malformed-tool-calls.test.ts index 5b8e867c1..a2e0af0fe 100644 --- a/packages/ai/test/transform-messages-malformed-tool-calls.test.ts +++ b/packages/ai/test/transform-messages-malformed-tool-calls.test.ts @@ -272,3 +272,25 @@ describe("transformMessages drops malformed (empty-name) tool calls", () => { expect(toolResults[0]?.toolName).toBe("read"); }); }); + +describe("transformMessages drops assistant images from provider replay", () => { + it("preserves replayable text while removing native image artifacts", () => { + const messages: Message[] = [ + { role: "user", content: "Draw a dot", timestamp: 1 }, + assistant( + [ + { type: "text", text: "Here it is." }, + { type: "image", data: "aW1hZ2U=", mimeType: "image/png" }, + ], + 2, + ), + ]; + + const transformed = transformMessages(messages, model); + + expect(transformed[1]).toMatchObject({ + role: "assistant", + content: [{ type: "text", text: "Here it is." }], + }); + }); +}); diff --git a/packages/ai/test/transform-messages-redact-sensitive.test.ts b/packages/ai/test/transform-messages-redact-sensitive.test.ts new file mode 100644 index 000000000..590bb3b8d --- /dev/null +++ b/packages/ai/test/transform-messages-redact-sensitive.test.ts @@ -0,0 +1,200 @@ +import { describe, expect, it } from "bun:test"; +import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; +import type { AssistantMessage, Message, Model, ToolCall, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +function makeModel(): Model<"openai-responses"> { + return buildModel({ + api: "openai-responses", + name: "GPT Test", + id: "gpt-test", + provider: "openai", + baseUrl: "https://api.openai.com/v1", + contextWindow: 8192, + maxTokens: 2048, + input: ["text"], + reasoning: false, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + }); +} + +describe("transformMessages redact sensitive credentials", () => { + it("redacts already-masked and real tokens from outbound messages", () => { + const messages: Message[] = [ + { + role: "user", + content: "Token: gho_************************************", + timestamp: Date.now(), + }, + { + role: "assistant", + content: [ + { + type: "text", + text: "I found this key: sk-proj-************************************", + }, + { + type: "toolCall", + id: "call_x", + name: "bash", + arguments: { + command: "echo gho_************************************", + }, + }, + ], + api: "openai-responses", + provider: "openai", + model: "gpt-test", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: Date.now(), + }, + { + role: "toolResult", + toolCallId: "call_x", + toolName: "bash", + content: [{ type: "text", text: "Token is ghp_************************************ inside output" }], + isError: false, + timestamp: Date.now(), + }, + ]; + + const transformed = transformMessages(messages, makeModel()); + + // 1. Verify user message is redacted + const userMsg = transformed[0]; + expect(userMsg.role).toBe("user"); + expect(userMsg.content).toBe("Token: [github_token_redacted]"); + + // 2. Verify assistant message text and toolCall arguments are redacted + const assistantMsg = transformed[1]; + expect(assistantMsg.role).toBe("assistant"); + const castAssistantMsg = assistantMsg as AssistantMessage; + const assistantContent = castAssistantMsg.content; + const textBlock = assistantContent[0]; + expect(textBlock.type).toBe("text"); + if (textBlock.type === "text") { + expect(textBlock.text).toBe("I found this key: [openai_token_redacted]"); + } + + const toolCallBlock = assistantContent[1]; + expect(toolCallBlock.type).toBe("toolCall"); + + // 3. Verify toolResult message is redacted + const resultMsg = transformed[2]; + expect(resultMsg.role).toBe("toolResult"); + const toolResultMsg = resultMsg as ToolResultMessage; + const toolResultBlock = toolResultMsg.content[0]; + expect(toolResultBlock.type).toBe("text"); + if (toolResultBlock.type === "text") { + expect(toolResultBlock.text).toBe("Token is [github_token_redacted] inside output"); + } + if (toolCallBlock.type === "toolCall") { + const toolCall = toolCallBlock as ToolCall; + const commandArg = toolCall.arguments?.command; + expect(commandArg).toBe("echo [github_token_redacted]"); + } + }); + + it("drops an Anthropic thinking signature when redacting its signed content", () => { + const model = buildModel({ + api: "anthropic-messages", + name: "Claude Test", + id: "claude-test", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + contextWindow: 8192, + maxTokens: 2048, + input: ["text"], + reasoning: true, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + }); + const messages: Message[] = [ + { + role: "assistant", + content: [ + { + type: "thinking", + thinking: "Use sk-ABCdef1234567890ABCdef1234567890ABCdef1234567890ABCdef123456.", + thinkingSignature: "signed-thinking-bytes", + }, + ], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-test", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: Date.now(), + }, + ]; + + const transformed = transformMessages(messages, model); + + expect(transformed[0]).toMatchObject({ role: "assistant", content: [] }); + }); + + it("drops a tool thought signature after redacting its arguments", () => { + const messages: Message[] = [ + { + role: "assistant", + content: [ + { + type: "toolCall", + id: "call_signed", + name: "run", + arguments: { token: "sk-ABCdef1234567890ABCdef1234567890ABCdef1234567890ABCdef123456" }, + thoughtSignature: "signed-tool-arguments", + }, + ], + api: "openai-responses", + provider: "openai", + model: "gpt-test", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: Date.now(), + }, + ]; + + const transformed = transformMessages(messages, makeModel()); + const block = (transformed[0] as AssistantMessage).content[0]; + + expect(block).toMatchObject({ + type: "toolCall", + arguments: { token: "[openai_token_redacted]" }, + }); + if (block.type === "toolCall") { + expect(block.thoughtSignature).toBeUndefined(); + } + }); + + it("preserves credential-shaped prose that is not a plausible live token", () => { + const lookalike = "sk-abcdefghijklmnopqrstuvwxyz"; + const transformed = transformMessages( + [{ role: "user", content: `The example key is ${lookalike}.`, timestamp: Date.now() }], + makeModel(), + ); + + expect(transformed[0]).toMatchObject({ role: "user", content: `The example key is ${lookalike}.` }); + }); +}); diff --git a/packages/ai/test/usage-attribution.test.ts b/packages/ai/test/usage-attribution.test.ts index 44d4c5dd8..d5533c942 100644 --- a/packages/ai/test/usage-attribution.test.ts +++ b/packages/ai/test/usage-attribution.test.ts @@ -21,6 +21,19 @@ const OPENAI_MODEL: Model<"openai-completions"> = buildModel({ maxTokens: 8_192, }); +const OPENROUTER_MODEL: Model<"openai-completions"> = buildModel({ + id: "deepseek/deepseek-v4-flash", + name: "DeepSeek V4 Flash", + api: "openai-completions", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + reasoning: true, + input: ["text"], + cost: { input: 0.098, output: 0.196, cacheRead: 0.02, cacheWrite: 0 }, + contextWindow: 1_048_576, + maxTokens: 384_000, +}); + function blankUsage(): Usage { return { input: 0, @@ -54,6 +67,21 @@ describe("openai-completions parseChunkUsage", () => { expect(usage.reasoningTokens).toBe(40); }); + it("uses OpenRouter's reported account charge instead of the catalog estimate", () => { + const usage = parseChunkUsage( + { + prompt_tokens: 1_000_000, + completion_tokens: 100_000, + cost: 0.42, + }, + OPENROUTER_MODEL, + undefined, + ); + + expect(usage.cost.total).toBe(0.42); + expect(usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite).toBeCloseTo(0.42); + }); + it("omits reasoningTokens when no reasoning_tokens are reported", () => { const usage = parseChunkUsage({ prompt_tokens: 50, completion_tokens: 25 }, OPENAI_MODEL, undefined); diff --git a/packages/ai/test/xai-oauth-usage.test.ts b/packages/ai/test/xai-oauth-usage.test.ts new file mode 100644 index 000000000..7647974e8 --- /dev/null +++ b/packages/ai/test/xai-oauth-usage.test.ts @@ -0,0 +1,177 @@ +import { describe, expect, it } from "bun:test"; +import { buildXAICliBillingUrl } from "@oh-my-pi/pi-ai/oauth/xai-oauth"; +import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; +import type { UsageFetchParams } from "@oh-my-pi/pi-ai/usage"; +import { xaiOauthUsageProvider } from "@oh-my-pi/pi-ai/usage/xai-oauth"; + +const USER_ID = "cf12ecb5-cca4-4ba0-9f02-298071a2d052"; + +const accessTokenFixture = (() => { + const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url"); + const body = Buffer.from(JSON.stringify({ sub: USER_ID })).toString("base64url"); + return `${header}.${body}.sig`; +})(); + +function makeBillingPayload(overrides?: Record) { + const periodEnd = new Date(Date.now() + 2 * 24 * 60 * 60 * 1000).toISOString(); + const periodStart = new Date(Date.now() - 5 * 24 * 60 * 60 * 1000).toISOString(); + return { + config: { + creditUsagePercent: 18, + currentPeriod: { + end: periodEnd, + start: periodStart, + type: "USAGE_PERIOD_TYPE_WEEKLY", + }, + productUsage: [ + { product: "GrokBuild", usagePercent: 16 }, + { product: "Api", usagePercent: 2 }, + ], + ...overrides, + }, + }; +} + +function makeCredential(overrides?: Partial): UsageFetchParams["credential"] { + return { + type: "oauth", + accessToken: accessTokenFixture, + refreshToken: "refresh-fixture", + expiresAt: Date.now() + 3_600_000, + ...overrides, + }; +} + +function capturingFetch(payload: unknown): { + fetch: FetchImpl; + calls: Array<{ url: string; headers: Record; redirect?: RequestInit["redirect"] }>; +} { + const calls: Array<{ url: string; headers: Record; redirect?: RequestInit["redirect"] }> = []; + const fetch: FetchImpl = async (input, init) => { + const headers: Record = {}; + const raw = init?.headers; + if (raw && typeof raw === "object" && !Array.isArray(raw)) { + for (const [key, value] of Object.entries(raw as Record)) { + headers[key.toLowerCase()] = value; + } + } + const url = String(input); + calls.push({ url, headers, redirect: init?.redirect }); + if (url.includes("/oauth2/userinfo")) { + return new Response(JSON.stringify({ sub: USER_ID, email: "user@example.com" }), { + status: 200, + headers: { "content-type": "application/json" }, + }); + } + return new Response(JSON.stringify(payload), { + status: 200, + headers: { "content-type": "application/json" }, + }); + }; + return { fetch, calls }; +} + +describe("xai-oauth usage provider", () => { + it("accepts stored OAuth credentials but never shared API-key fallbacks", () => { + expect(xaiOauthUsageProvider.supports?.({ provider: "xai-oauth", credential: makeCredential() })).toBe(true); + expect( + xaiOauthUsageProvider.supports?.({ + provider: "xai-oauth", + credential: { type: "api_key", apiKey: accessTokenFixture }, + }), + ).toBe(false); + }); + + it("maps weekly credit and product usage with CLI-aligned billing headers", async () => { + const { fetch, calls } = capturingFetch(makeBillingPayload()); + const report = await xaiOauthUsageProvider.fetchUsage( + { provider: "xai-oauth", credential: makeCredential() }, + { fetch: fetch }, + ); + + expect(report?.limits.map(limit => limit.id)).toEqual([ + "xai-oauth:credits:1w", + "xai-oauth:product:grokbuild:1w", + "xai-oauth:product:api:1w", + ]); + expect(report?.limits[0]?.amount.usedFraction).toBeCloseTo(0.18, 5); + expect(report?.metadata?.accountId).toBe(USER_ID); + expect(report?.metadata?.email).toBe("user@example.com"); + + const billingCall = calls.find(call => call.url.includes("/v1/billing")); + expect(billingCall?.url).toBe(buildXAICliBillingUrl()); + expect(billingCall?.headers).toEqual({ + authorization: `Bearer ${accessTokenFixture}`, + accept: "application/json", + "x-xai-token-auth": "xai-grok-cli", + }); + expect(billingCall?.redirect).toBe("error"); + }); + + it("uses a stored email without an extra userinfo request", async () => { + const { fetch, calls } = capturingFetch(makeBillingPayload()); + const report = await xaiOauthUsageProvider.fetchUsage( + { + provider: "xai-oauth", + credential: makeCredential({ accountId: "stored-account", email: "stored@example.com" }), + }, + { fetch: fetch }, + ); + + expect(report?.metadata?.accountId).toBe("stored-account"); + expect(report?.metadata?.email).toBe("stored@example.com"); + expect(calls.some(call => call.url.includes("/oauth2/userinfo"))).toBe(false); + }); + + it("maps a positive on-demand cap", async () => { + const report = await xaiOauthUsageProvider.fetchUsage( + { + provider: "xai-oauth", + credential: makeCredential(), + }, + { fetch: capturingFetch(makeBillingPayload({ onDemandCap: { val: 50 }, onDemandUsed: { val: 10 } })).fetch }, + ); + + const onDemand = report?.limits.find(limit => limit.id === "xai-oauth:on-demand"); + expect(onDemand?.amount.used).toBe(10); + expect(onDemand?.amount.limit).toBe(50); + expect(onDemand?.amount.usedFraction).toBeCloseTo(0.2, 5); + }); + + it("still reports usage when the weekly period has just ended", async () => { + const periodEnd = new Date(Date.now() - 60_000).toISOString(); + const periodStart = new Date(Date.now() - 8 * 24 * 60 * 60 * 1000).toISOString(); + const report = await xaiOauthUsageProvider.fetchUsage( + { provider: "xai-oauth", credential: makeCredential() }, + { + fetch: capturingFetch( + makeBillingPayload({ + currentPeriod: { + end: periodEnd, + start: periodStart, + type: "USAGE_PERIOD_TYPE_WEEKLY", + }, + }), + ).fetch, + }, + ); + + expect(report?.limits[0]?.id).toBe("xai-oauth:credits:1w"); + expect(report?.limits[0]?.window?.resetsAt).toBe(Date.parse(periodEnd)); + }); + + it("skips expired OAuth tokens and returns null for rejected billing", async () => { + const expired = await xaiOauthUsageProvider.fetchUsage( + { provider: "xai-oauth", credential: makeCredential({ expiresAt: Date.now() - 1 }) }, + { fetch: capturingFetch(makeBillingPayload()).fetch }, + ); + expect(expired).toBeNull(); + + const denied: FetchImpl = async () => new Response("denied", { status: 403 }); + const report = await xaiOauthUsageProvider.fetchUsage( + { provider: "xai-oauth", credential: makeCredential() }, + { fetch: denied }, + ); + expect(report).toBeNull(); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 7124fdc93..54026822e 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,118 @@ ## [Unreleased] +### Changed + +- Renamed `codex-auto-review` model to `GPT-5.3 Codex Spark` with updated pricing and capabilities +- Removed image input support from GPT-5.3 Codex Spark (text-only) +- Reduced GPT-5.3 Codex Spark context window from 272K to 128K tokens +- Changed GPT-5.3 Codex Spark thinking efforts from `["minimal", "low", "medium", "high", "xhigh"]` to `["low", "medium", "high", "xhigh"]` +- Updated pricing for multiple AI models across providers (costs adjusted in models.json) +- Reduced max output tokens for an unspecified model from 16384 to 8192 +- Added image input support to Venice AI text model + +## [17.0.8] - 2026-07-22 + +### Added + +- Added support for several new models across multiple providers, including MiniMax M3, Gemini 3.5 Flash Lite, Gemini 3.6 Flash (with thinking support), Hy3, Doubao-Seed-Character, LongCat 2.0, Laguna S 2.1 (free and paid tiers), Qwen 3.6 35B A3B, SWE-1.6 Slow (devin agent catalog), and XiaomiMiMo/MiMo-V2.5. + +### Changed + +- Updated Grok 4.5 API type to "openai-responses" and updated "o3-mini" to support thinking capabilities with the "kimi" thinking format. +- Renamed OpenRouter-specific models and routers to include "OpenRouter" in their names (e.g., "OpenRouter Auto Router (Beta)", "OpenRouter Body Builder (beta)", and "OpenRouter Pareto Code Router"). +- Updated context window sizes, costs, and token limits for numerous models. + +### Fixed + +- Fixed an issue where GPT-5.6 Codex SKUs lost usable context window capacity due to dynamic discovery values overwriting bundled limits. +- Fixed OpenAI Codex discovery dropping account-listed ChatGPT-only models (such as GPT-5.3 Codex Spark) when they are unavailable through the public API. +- Fixed Codex catalog discovery hiding models when multiple OAuth accounts are configured by independently fetching and merging catalogs from all accounts. +- Fixed cached models reusing a bundled request model (such as GitHub Copilot long-context variants) being incorrectly flagged as unrestorable and dropped after a restart. +- Fixed LM Studio discovery reporting a model's theoretical maximum context length instead of the actual loaded context window size of the running instance. + +### Removed + +- Removed several deprecated model families from the devin catalog, including Claude Fable 5, Claude Opus 4.6/4.7, Claude Sonnet 4.6/5, DeepSeek V4 Pro, Gemini 3.1 Pro, Gemini 3.5 Flash, GLM-5.2, SWE-1.6, and Nemotron 3 Ultra. +- Removed GPT-5 through GPT-5.3 Codex variants and GPT-5.4 nano from the openai-codex catalog. + +## [17.0.6] - 2026-07-20 + +### Added + +- Added static fallback seed for Devin's `swe-1-7` model so it is bundled even when catalog generation runs without a Devin session token. + +### Fixed + +- Collapsed Devin's six GLM-5.2 variants into two logical entries (`glm-5-2` for 200K free, `glm-5-2-1m` for 1M paid). The 200K entry routes every thinking effort to the free `glm-5-2` wire UID — never to the quota-gated `glm-5-2-max` or `glm-5-2-none` — so GLM-5.2 works even when the weekly usage quota is exhausted. + +## [17.0.5] - 2026-07-18 + +### Added + +- Added an Anthropic compatibility flag to allow non-official OAuth endpoints to opt into configured Claude Code fingerprint header overrides. + +### Fixed + +- Fixed a security issue where sensitive provider-defined request headers (such as API keys or credentials) were serialized in plaintext within the model cache (models.db). The cache now omits these headers, securely invalidates older cached rows, and restores or refetches them dynamically. +- Fixed OpenAI Codex discovery to respect caller-supplied fetch configurations (such as proxies or custom CAs) and correctly replace stale bundled models with the authenticated account catalog. +- Fixed stream timeouts and retry loops during long prefills on local loopback or RFC1918 backends (such as litellm proxies fronting local servers) by applying the local stream-timeout floor to these backends. +- Fixed Kimi K3 models served through generic OpenAI-compatible routes exposing unsupported reasoning efforts instead of the mandatory low/high/max scale. + +## [17.0.4] - 2026-07-18 + +### Changed + +- Kimi-family models now use MFJS tool schema on all hosts, including proxies like OpenRouter that forward schemas to Moonshot + +## [17.0.3] - 2026-07-17 + +### Fixed + +- Logged LiteLLM rich-metadata endpoint failures once with their endpoint and status before falling back to incomplete `/v1/models` data ([#5801](https://github.com/can1357/oh-my-pi/issues/5801)). +- Fixed authenticated Kimi Code discovery to preserve live effort levels, default effort, mandatory-thinking state, and per-model protocol metadata ([#5893](https://github.com/can1357/oh-my-pi/issues/5893)). +- Fixed LiteLLM provider ignoring per-model pricing: `mapLiteLLMRichEntry` now reads `input_cost_per_token` / `output_cost_per_token` (plus cache costs) from LiteLLM rich metadata and maps them to `cost.input` / `cost.output`, falling back to the bundled reference only when LiteLLM omits cost, so proxied models no longer display as free ([#5818](https://github.com/can1357/oh-my-pi/issues/5818)). + +## [17.0.2] - 2026-07-17 + +### Changed + +- Increased the maximum output tokens (maxTokens) from 32,768 to 65,536 for Kimi K2.7-Code models on Fireworks. + +### Fixed + +- Fixed a regression where the context window for openai-codex GPT-5.6 models (Luna, Sol, Terra) incorrectly fell back to 272,000 instead of preserving its 372,000 capacity. +- Fixed Umans PAYG models incorrectly displaying as "Free" in /models by correctly sourcing their published per-token rates. +- Fixed native moonshot/kimi-k3 capabilities and pricing, ensuring it correctly reflects its official pricing, 1M context window, image input support, reasoning capabilities, and 128k output token limit. + +## [17.0.1] - 2026-07-16 + +### Added + +- Added GPT-5.6 Luna, Sol, and Terra entries for Amazon Bedrock, Azure, and Cloudflare +- Added KAT-Coder Air/Pro V2.5 entries across Kilo, OpenRouter, NanoGPT, and Vercel +- Added Inkling model entries for Baseten and Vercel AI Gateway +- Added Umans DeepSeek V4 Pro DSpark as an experimental model listing +- Added Claude Opus 4.7 Fast and 4.8 Fast on Vercel AI Gateway +- Added Workers AI GLM-5.2, Muse Spark 1.1, Stealth GPT-5.6 Sol, and nano-gpt-help entries + +### Changed + +- Added image input and reasoning support to several existing Codeium and Kilo GPT-5.6 models +- Enabled image input and reasoning for Gemini Flash Latest and Grok 4.5 +- Renamed many model labels for consistency, including Claude, Grok, DeepSeek, GLM, and Gemi­ni names +- Updated pricing for many existing models, including input, output, and cache cost values +- Updated context window and max token limits for many catalog models across providers + +### Fixed + +- Fixed Z.AI (GLM) coding-plan token costs all showing as "Free" in `/models`: the `zai` provider descriptor sourced the models.dev `zai-coding-plan` key (all-$0 subscription rates) instead of the `zai` pay-as-you-go key, which carries the real per-token rates for the identical GLM ids ([#5598](https://github.com/can1357/oh-my-pi/issues/5598)). +- Fixed custom Anthropic endpoints receiving the first-party-only `eager_input_streaming` tool field by default ([#5572](https://github.com/can1357/oh-my-pi/issues/5572)). +- Added resolved OpenAI sampling-parameter compatibility metadata for o-series and GPT-5+ models. +- Fixed GitHub Copilot `mai-code-1-flash-picker` (and other `mai-*` models) to route through the `/responses` endpoint instead of `/chat/completions`, which rejected them with `400 unsupported_api_for_model` ([#5612](https://github.com/can1357/oh-my-pi/issues/5612)). +- Extended the reasoning `streamIdleTimeoutMs` floor (300s) to native Kimi K2.7 Code (`kimi-k2.7-code` / `kimi-k2.7-code-highspeed`), which previously fell through to the 120s default and aborted on long reasoning turns ([#4836](https://github.com/can1357/oh-my-pi/issues/4836)). +- Fixed GLM-5.x coding-plan streams via the OpenCode Go/Zen gateways (`opencode.ai/zen/…`) timing out with `OpenAI completions stream stalled while waiting for the next event` during slow plan-writing/reasoning phases. The 600s idle-timeout floor for GLM coding-plan SKUs was gated to the native Z.AI/Zhipu hosts only, so OpenCode-fronted GLM fell back to the 120s default watchdog. ([#4758](https://github.com/can1357/oh-my-pi/issues/4758)) + ## [16.5.2] - 2026-07-14 ### Fixed diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 5ea868aed..72c9aa6c8 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "17.0.0", + "version": "17.0.8", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 1d9df8b78..09c65d17a 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -231,6 +231,30 @@ function hasBillableCost(cost: ModelSpec["cost"]): boolean { return cost.input !== 0 || cost.output !== 0 || cost.cacheRead !== 0 || cost.cacheWrite !== 0; } +function applyUmansPricingFallback(models: readonly ModelSpec[], modelsDevModels: readonly ModelSpec[]): ModelSpec[] { + const paygCosts = new Map(); + for (const model of modelsDevModels) { + if (model.provider === "umans" && hasBillableCost(model.cost)) { + paygCosts.set(model.id, model.cost); + } + } + + // The public endpoint exposes this technical alias for Umans Flash, but + // models.dev publishes pricing only for the recommended `umans-flash` id. + const flashCost = paygCosts.get("umans-flash"); + if (flashCost) { + paygCosts.set("umans-qwen3.6-35b-a3b", flashCost); + } + + return models.map(model => { + if (model.provider !== "umans" || hasBillableCost(model.cost)) { + return model; + } + const cost = paygCosts.get(model.id); + return cost ? { ...model, cost: { ...cost } } : model; + }); +} + function applyCodexPricingFallback(models: readonly ModelSpec[]): ModelSpec[] { const openAIModels = new Map( models @@ -524,19 +548,25 @@ async function generateModels() { allModels.push(...buildFireworksFastSeed()); const specialDiscoverySources = [ - { label: "Antigravity", fetch: fetchAntigravityModels }, - { label: "Codex", fetch: fetchCodexDiscoveryModels }, + { label: "Antigravity", providerId: "google-antigravity", authoritative: false, fetch: fetchAntigravityModels }, + { label: "Codex", providerId: "openai-codex", authoritative: true, fetch: fetchCodexDiscoveryModels }, ] as const; const specialDiscoveries = await Promise.all( specialDiscoverySources.map(async source => ({ label: source.label, + providerId: source.providerId, + authoritative: source.authoritative, models: await source.fetch(), })), ); + const authoritativeSpecialDiscoveryProviders = new Set(); for (const discovery of specialDiscoveries) { if (discovery.models.length > 0) { console.log(`Added ${discovery.models.length} models from ${discovery.label} discovery`); allModels.push(...discovery.models); + if (discovery.authoritative) { + authoritativeSpecialDiscoveryProviders.add(discovery.providerId); + } } } @@ -563,6 +593,7 @@ async function generateModels() { !DISCOVERY_ONLY_PROVIDERS.has(model.provider) && !RETIRED_PROVIDERS.has(model.provider) && !authoritativeCatalogProviders.has(model.provider) && + !authoritativeSpecialDiscoveryProviders.has(model.provider) && !modelsDevSnapshotExcludedProviders.has(model.provider) ) { allModels.push(model); @@ -571,6 +602,7 @@ async function generateModels() { } allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels); + allModels = applyUmansPricingFallback(allModels, modelsDevModels); allModels = applyPremiumMultiplierOverrides(allModels); allModels = applyCodexPricingFallback(allModels); allModels = applyKimiMaxTokensCap(allModels); diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index eb670b16e..7ebf256ec 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -362,4 +362,12 @@ function applyOpenAICatalogPolicy(model: ModelSpec, parsedModel: OpenAIMode model.contextWindow = 272000; } } + // GPT-5.6 luna/sol/terra on the Codex transport: OpenAI's Codex model + // registry declares context_window = max_context_window = 372000, but Codex + // discovery under-reports it — omitting the field for some accounts and + // actively returning 272000 for others (#5705, #6259). Pin the true 372K + // input window on the bundled catalog; discovery enforces the same floor. + if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.6")) { + model.contextWindow = 372000; + } } diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index fafb6cac9..71f7de8d3 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -109,7 +109,8 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res signingEndpoint, disableStrictTools: isAzure, disableAdaptiveThinking: false, - supportsEagerToolInputStreaming: !isCopilot, + allowAnthropicHeaderOverrides: false, + supportsEagerToolInputStreaming: official, // Long cache retention is only sent to the official API by default; // proxies opt in explicitly via `compat.supportsLongCacheRetention: true`. supportsLongCacheRetention: official, diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 2425a4744..7d21a3ead 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -15,9 +15,11 @@ import { isDeepseekModelIdOrName, isGlm52ReasoningEffortModelId, isGrokReasoningEffortCapable, + isKimiK3ModelId, isKimiK26ModelId, isKimiModelId, isMimoModelIdOrName, + isOpenAISamplingRestrictedModelId, isQwenModelId, modelFamilyToken, } from "../identity/family"; @@ -37,8 +39,8 @@ const GLM_CODING_PLAN_MODEL_PATTERN = /(^|\/)glm-5(?:[.-]|$)/i; const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; /** Direct DeepSeek reasoning models stall between thinking and answer phases. */ const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; -/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */ -const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; +/** Kimi K2.6 and native K2.7 Code can spend several minutes reasoning before the first visible token. */ +const KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; /** * Native Kimi K2.7 Code requires `thinking.type: "enabled"` and rejects * disabled thinking. Match the public id, its Fast variant, and the @@ -84,6 +86,7 @@ function resolveReasoningDisableMode( case "openrouter": return "openrouter-enabled-false"; case "zai": + case "kimi": return "zai-thinking-disabled"; case "qwen": return "qwen-enable-thinking-false"; @@ -151,14 +154,32 @@ const OPENCODE_WHEN_THINKING: NonNullable = { reasoningContentField: "reasoning_content", }; +const KIMI_K3_REASONING_EFFORT_MAP: NonNullable = { + minimal: "low", + medium: "high", + xhigh: "max", + max: "max", +}; + const MIMO_REASONING_EFFORT_MAP: NonNullable = { minimal: "low", xhigh: "high", }; -function mergeMimoReasoningEffortMap(compat: ResolvedOpenAISharedCompat, enabled: boolean): void { - if (!enabled) return; - compat.reasoningEffortMap = { ...MIMO_REASONING_EFFORT_MAP, ...compat.reasoningEffortMap }; +function mergeModelReasoningEffortMap( + compat: ResolvedOpenAISharedCompat, + modelId: string, + isMimoReasoningEffortModel: boolean, +): void { + let detected: NonNullable; + if (isKimiK3ModelId(modelId)) { + detected = KIMI_K3_REASONING_EFFORT_MAP; + } else if (isMimoReasoningEffortModel) { + detected = MIMO_REASONING_EFFORT_MAP; + } else { + return; + } + compat.reasoningEffortMap = { ...detected, ...compat.reasoningEffortMap }; } function detectStrictModeSupport(provider: string, baseUrl: string): boolean { @@ -246,6 +267,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const isKimiModel = isKimiModelId(spec.id); const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative"); const isMoonshotKimi = isKimiModel && isMoonshotNative; + // Native Kimi K3 uses OpenAI-style `reasoning_effort` with mandatory + // low/high/max thinking, not the K2.x binary `thinking: { type }` block. + const isKimiK3 = isKimiK3ModelId(spec.id); + const isMoonshotKimiK3 = isMoonshotKimi && isKimiK3; const requiresEnabledThinking = isMoonshotKimi && matchesKimiK27CodeFamily(spec); const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id); const isAnthropicModel = @@ -298,6 +323,14 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const isLocalOpenAICompatBackend = !PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) && (LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)); + // Stream-timeout floor applies to ANY loopback/RFC1918 backend, INCLUDING + // local proxies (litellm) excluded from `isLocalOpenAICompatBackend` above: + // widening the first-event/idle abort ceiling only helps a slow local + // upstream and never pushes an extra wire field, so the proxy carve-out (a + // `replayReasoningContent` safety measure) must not also strip the timeout + // floor. Without this, a loopback litellm fronting a cold/reprocessing + // llama-server aborts prefill at the 100s default and retry-loops (#4786). + const isLocalServingBackend = isLocalOpenAICompatBackend || hasLocalLoopbackBaseUrl(baseUrl); const useMaxTokens = isMistral || @@ -357,17 +390,20 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv // for minutes while reasoning or cold-loading weights; widen the idle // timeout so warm-ups stop aborting and retrying. const streamIdleTimeoutMs = - GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu) + GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu || isOpenCodeHost) ? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS : provider === "alibaba-coding-plan" ? ALIBABA_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS : isXiaomiMimo ? XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS - : spec.reasoning && isKimiK26ModelId(spec.id) - ? KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS + : spec.reasoning && + (isKimiK26ModelId(spec.id) || + isMoonshotKimiK3 || + (isMoonshotKimi && matchesKimiK27CodeFamily(spec))) + ? KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS : spec.reasoning && isDirectDeepseekApi ? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS - : isLocalOpenAICompatBackend + : isLocalServingBackend ? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS : undefined; @@ -384,7 +420,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv ? "openrouter" : "raw"; const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = - isZai || isZhipu || isMoonshotKimi || isXiaomiMimo + (isMoonshotKimi && !isMoonshotKimiK3) || isZai || isZhipu || isXiaomiMimo ? "zai" : isOpenRouter ? "openrouter" @@ -409,7 +445,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv supportsReasoningEffort: !isGrok && !isXiaomiMimo && (!(isZai || isZhipu) || supportsZaiReasoningEffort), // GitHub Copilot's chat-completions endpoint rejects reasoning params wholesale. supportsReasoningParams: provider !== "github-copilot", - reasoningEffortMap: isMimoReasoningEffortModel ? MIMO_REASONING_EFFORT_MAP : {}, + // OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit + // temperature/top_p/… with a 400 on every serving host (#5606). + supportsSamplingParams: !isOpenAISamplingRestrictedModelId(spec.id), + reasoningEffortMap: {}, supportsUsageInStreaming: !isCerebras, // pi-ai's thinking-loop guard is gemini-only; default the flag from the // family classifier so OpenAI-compat proxies serving Gemini are covered. @@ -422,7 +461,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv // every call since the family can otherwise emit very long reasoning traces // before the final answer. alwaysSendMaxTokens: isKimiModel, - disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel, + // Native Kimi K3 always reasons through `reasoning_effort` (never the + // K2.x binary `thinking` block that #827's forced-tool-choice conflict is + // about), so suppressing its effort would leave K3 in an unsupported mode. + disableReasoningOnForcedToolChoice: (isKimiModel && !isMoonshotKimiK3) || isAnthropicModel, disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter, supportsToolChoice: !isDirectDeepseekReasoning, supportsForcedToolChoice: !requiresEnabledThinking, @@ -432,16 +474,16 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv requiresAssistantAfterToolResult: isMistral, requiresThinkingAsText: isMistral, requiresMistralToolIds: isMistral, - // Only Kimi's native hosts (Moonshot / Kimi-code, matched by `isMoonshotKimi`) - // speak the z.ai binary `thinking: { type }` field. Kimi reached through - // OpenAI-compatible proxies — Fireworks' Fire Pass router, OpenCode's gateway, - // etc. — drives reasoning via OpenAI-style `reasoning_effort` - // (low|medium|high|xhigh|max|none), so those stay on the "openai" path. + // Only Kimi's native K2.x hosts (Moonshot / Kimi-code, matched by + // `isMoonshotKimi`) speak the z.ai binary `thinking: { type }` field. + // K3 and Kimi reached through OpenAI-compatible proxies drive reasoning + // via OpenAI-style `reasoning_effort`. // NVIDIA NIM hosts Qwen with the vLLM convention // (`chat_template_kwargs.enable_thinking`); top-level `enable_thinking` // is rejected by NIM's `additionalProperties: false` request schema // (issue #2299). thinkingFormat, + kimiApiFormat: undefined, reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat), omitReasoningEffort: false, includeEncryptedReasoning: true, @@ -512,7 +554,11 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv supportsStrictMode: detectStrictModeSupport(provider, baseUrl), extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined, toolStrictMode: isCerebras ? "all_strict" : "mixed", - toolSchemaFlavor: isMoonshotNative ? "moonshot-mfjs" : undefined, + // Kimi-family ids trigger MFJS on any host, not just native base URLs: + // proxies (OpenRouter, custom gateways) forward `tools.function.parameters` + // to Moonshot verbatim, which 400s on non-MFJS constructs. + toolSchemaFlavor: + isMoonshotNative || isKimiModel ? "moonshot-mfjs" : isLocalOpenAICompatBackend ? "grammar" : undefined, streamIdleTimeoutMs, stripDeepseekSpecialTokens: isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"), @@ -534,7 +580,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; } - mergeMimoReasoningEffortMap(compat, isMimoReasoningEffortModel); + mergeModelReasoningEffortMap(compat, spec.id, isMimoReasoningEffortModel); const whenThinkingPolicy = spec.compat?.whenThinking ?? (isOpenCodeProvider && spec.reasoning ? OPENCODE_WHEN_THINKING : undefined); @@ -547,7 +593,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv if (whenThinkingPolicy.omitReasoningEffort === undefined && !variant.supportsReasoningEffort) { variant.omitReasoningEffort = true; } - mergeMimoReasoningEffortMap(variant, isMimoReasoningEffortModel); + mergeModelReasoningEffortMap(variant, spec.id, isMimoReasoningEffortModel); compat.whenThinking = variant; } @@ -583,9 +629,13 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol const isAnthropicModel = id ? isClaudeModelId(id) || isAnthropicNamespacedModelId(id) : false; const isDeepseekFamily = id ? isDeepseekModelIdOrName(id) || isDeepseekModelIdOrName(spec.name) : false; const reasoningCapable = Boolean(spec.reasoning); - const isLocalOpenAICompatBackend = - !PROXY_OPENAI_COMPAT_PROVIDERS.has(spec.provider) && - (LOCAL_OPENAI_COMPAT_PROVIDERS.has(spec.provider) || hasLocalLoopbackBaseUrl(baseUrl)); + // `replayReasoningContent` is Responses-only-false, so the proxy carve-out is + // irrelevant here; the stream-timeout floor still applies to ANY loopback / + // RFC1918 backend, including local proxies (litellm), so a slow local + // upstream is not aborted at the 100s default and retry-looped (#4786). + const isLocalServingBackend = + (!PROXY_OPENAI_COMPAT_PROVIDERS.has(spec.provider) && LOCAL_OPENAI_COMPAT_PROVIDERS.has(spec.provider)) || + hasLocalLoopbackBaseUrl(baseUrl); const compat: ResolvedOpenAIResponsesCompat = { supportsDeveloperRole: isAzure || isOpenAIUrl || hostMatchesUrl(baseUrl, "githubCopilot"), @@ -604,6 +654,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol spec.provider !== "xai-oauth" && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"), reasoningEffortMap: {}, supportsReasoningParams: true, + // OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit + // temperature/top_p/… with a 400 on every serving host (#5606). + supportsSamplingParams: !isOpenAISamplingRestrictedModelId(id), thinkingFormat, reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat), omitReasoningEffort: false, @@ -635,6 +688,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol openRouterRouting: undefined, isOpenRouterHost: isOpenRouter, wireModelIdMode: isOpenRouter ? "openrouter" : "raw", + // Mirrors buildOpenAICompat: Kimi behind a Responses-capable proxy still + // lands on Moonshot's MFJS validator. + toolSchemaFlavor: isKimiModel ? "moonshot-mfjs" : undefined, alwaysSendMaxTokens: spec.id ? isKimiModelId(spec.id) : false, enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini", supportsObfuscationOptOut: isOpenAIUrl || spec.provider === "openai", @@ -646,7 +702,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol emptyLengthFinishIsContextError: spec.provider === "ollama", usesOpenAIToolCallIdLimit: spec.provider === "openai", promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined, - streamIdleTimeoutMs: isLocalOpenAICompatBackend + streamIdleTimeoutMs: isLocalServingBackend ? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS : spec.compat?.streamIdleTimeoutMs, }; diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 33dad7763..4f0625fc5 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -1,11 +1,20 @@ import { type } from "arktype"; -import type { ModelSpec } from "../types"; +import { parseKnownModel, semverEqual } from "../identity/classify"; +import type { FetchImpl, ModelSpec } from "../types"; import { discoveryFetch } from "../utils"; import { CODEX_BASE_URL, CODEX_CLIENT_VERSION, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../wire/codex"; const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const; const DEFAULT_CONTEXT_WINDOW = 272_000; const DEFAULT_MAX_TOKENS = 128_000; +/** + * GPT-5.6 luna/sol/terra hard context capacity. OpenAI's Codex model registry + * declares context_window = max_context_window = 372000 (#5705), but Codex + * discovery under-reports it — omitting the field for some accounts and + * actively returning 272000 for others (#6259). Applied as a floor for these + * SKUs so the reported/absent value never regresses the real window. + */ +const GPT_5_6_CONTEXT_WINDOW = 372_000; const CODEX_REMOTE_COMPACTION = { enabled: true, api: "openai-codex-responses", @@ -24,7 +33,7 @@ const codexModelEntrySchema = type({ "default_reasoning_level?": "unknown", "supported_reasoning_levels?": "unknown", "input_modalities?": "unknown", - "supported_in_api?": "unknown", + "visibility?": "unknown", "priority?": "unknown", "prefer_websockets?": "unknown", "use_responses_lite?": "unknown", @@ -60,7 +69,7 @@ export interface CodexModelDiscoveryOptions { /** Abort signal for network request cancellation. */ signal?: AbortSignal; /** Optional fetch implementation override for tests. */ - fetchFn?: typeof fetch; + fetchFn?: FetchImpl; } /** @@ -208,13 +217,24 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo return null; } - const supportedInApi = toBoolean(payload.supported_in_api); - if (supportedInApi === false) { + const visibility = toNonEmptyString(payload.visibility)?.toLowerCase(); + if (visibility === "hide" || visibility === "hidden") { return null; } const name = toNonEmptyString(payload.display_name) ?? slug; - const contextWindow = toPositiveInt(payload.context_window) ?? DEFAULT_CONTEXT_WINDOW; + // GPT-5.6 luna/sol/terra have a 372000 hard window, but Codex discovery + // under-reports it: for some accounts the field is omitted, for others it is + // actively returned as 272000 (#6259). Treat GPT_5_6_CONTEXT_WINDOW as a + // floor for these SKUs so neither the omission nor the active under-report + // regresses the real capacity; other models honor the reported value with + // the generic 272000 fallback. + const parsed = parseKnownModel(slug); + const isGpt56 = parsed.family === "openai" && semverEqual(parsed.version, "5.6"); + const reportedContextWindow = toPositiveInt(payload.context_window); + const contextWindow = isGpt56 + ? Math.max(GPT_5_6_CONTEXT_WINDOW, reportedContextWindow ?? 0) + : (reportedContextWindow ?? DEFAULT_CONTEXT_WINDOW); const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow); const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels); const input = normalizeInputModalities(payload.input_modalities); diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index aa206cd9f..3eec81b7e 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -41,6 +41,16 @@ export const isKimiK26ModelId = memo((modelId: string): boolean => { return /(^|\/)kimi-k2(?:\.6|p6)(?:[-:]|$)/i.test(modelId); }); +/** + * Kimi K3 in any namespace form (`kimi-k3`, `kimi-k3.1`, `kimi-k3-turbo`, + * `moonshotai/kimi-k3`). K3 always reasons and drives thinking via OpenAI-style + * `reasoning_effort: "max"`, not the K2.x binary `thinking: { type }` block — + * see the moonshot discovery mapper and `buildOpenAICompat`. + */ +export const isKimiK3ModelId = memo((modelId: string): boolean => { + return /(^|\/)kimi-k3(?:\.\d+)?(?:[-.:_]|$)/i.test(modelId); +}); + /** * Claude ids in any namespace form: bare (`claude-*`), path-namespaced * (`anthropic/claude.x`), or dot-prefixed (`us.anthropic.claude-…`, @@ -163,6 +173,32 @@ export const supportsAllTurnsReasoningContext = isOpenAIWireGen54Plus; */ export const supportsCodexReasoningSummary = isOpenAIWireGen54Plus; +/** OpenAI proprietary reasoning families keyed off the parsed gpt version (gpt-5+). */ +const isOpenAIWireGen5Plus = memo((modelId: string): boolean => { + const parsed = parseOpenAIModel(bareModelId(modelId)); + if (!parsed) return false; + return semverGte(parsed.version, "5"); +}); + +/** o-series reasoning ids (`o1`, `o1-pro`, `o3`, `o3-mini`, `o4-mini`, `openai/o3`, …). */ +const O_SERIES_REASONING_RE = /(^|\/)o[134](?:[-.]|$)/i; + +/** + * OpenAI proprietary models whose serving path rejects explicit sampling + * parameters (`temperature`, `top_p`, `top_k`, …) with + * `400 Unsupported parameter: 'temperature' is not supported with this model`. + * Covers the o-series and the entire gpt-5+ generation — base, `mini`, `nano`, + * `codex*`, the `luna`/`sol`/`terra` SKUs, and the `-chat-latest` variants, + * since even the non-reasoning gpt-5 chat models reject sampling params (see + * litellm#13781). Holds regardless of which OpenAI-serving host proxies the + * model (official, Azure, GitHub Copilot). Version floor (not an allowlist) so + * 6.x inherits automatically. Issue #5606. + */ +export const isOpenAISamplingRestrictedModelId = memo((modelId: string): boolean => { + const bare = bareModelId(modelId); + return isOpenAIWireGen5Plus(modelId) || O_SERIES_REASONING_RE.test(bare); +}); + /** * Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` / * `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning diff --git a/packages/catalog/src/identity/reference.ts b/packages/catalog/src/identity/reference.ts index 461fc8e11..b795eec6b 100644 --- a/packages/catalog/src/identity/reference.ts +++ b/packages/catalog/src/identity/reference.ts @@ -8,7 +8,7 @@ * may strip `search`-style markers and prefers cache-pricing-complete * references, both of which would be wrong for canonical coalescing. */ -import type { Api, Model } from "../types"; +import type { Api, Model, ThinkingConfig } from "../types"; import { getBracketStrippedModelIdCandidates, getLongestModelLikeIdSegment, getModelLikeIdSegments } from "./id"; import { REFERENCE_TRAILING_MARKER_PATTERN } from "./markers"; @@ -137,8 +137,27 @@ function getReferenceCandidateIds(modelId: string): string[] { return [...candidates]; } +/** + * Inherit bundled reference thinking only for same-provider matches. Wire routing + * (`effortRouting`) is provider-specific; cross-provider inheritance can rewrite + * gateway ids (e.g. Portkey `@modal/GLM-5-2-FP8` → devin `glm-5-2`). + */ +export function inheritReferenceThinking( + modelThinking: ThinkingConfig | undefined, + reference: Pick, "provider" | "thinking"> | undefined, + provider: string, +): ThinkingConfig | undefined { + if (modelThinking !== undefined) return modelThinking; + if (!reference?.thinking) return undefined; + if (reference.provider !== provider) return undefined; + return reference.thinking; +} + /** Resolve a (possibly proxied/affixed) model id to its bundled upstream reference. */ export function resolveModelReference(modelId: string, index: ModelReferenceIndex): Model | undefined { + // Portkey/gateway wire ids (`@provider/model`) are opaque; fuzzy matching would + // map them to unrelated bundled entries (e.g. `@modal/GLM-5-2-FP8` → devin/glm-5-2). + if (modelId.startsWith("@")) return undefined; for (const candidate of getReferenceCandidateIds(modelId)) { const key = normalizeReferenceKey(candidate); const reference = index.exact.get(key) ?? index.suffixAlias.get(key); diff --git a/packages/catalog/src/model-cache.ts b/packages/catalog/src/model-cache.ts index 618996697..b426530aa 100644 --- a/packages/catalog/src/model-cache.ts +++ b/packages/catalog/src/model-cache.ts @@ -7,15 +7,21 @@ import { getModelDbPath } from "@oh-my-pi/pi-utils"; import type { Api, Model, ModelSpec } from "./types"; // Rows persist ModelSpec JSON (sparse `compat`, never the resolved record); -// the model manager rebuilds via `buildModel` on load. v8 invalidates Codex -// discovery rows predating provider-native V2 compaction metadata; v7 -// invalidated rows predating the Antigravity Gemini budget-mode migration -// (cached specs still carrying `thinking.mode: "google-level"` and the old -// 3.5-flash effort routing); v6 invalidated rows that may contain the retired -// unknown-limit sentinels (222222/8888); v5 invalidated rows predating +// the model manager rebuilds via `buildModel` on load. Request headers are +// intentionally omitted: arbitrary provider-defined header names can carry +// credentials. v10 deletes rows that may contain persisted headers and records +// which model ids lost headers and which cannot be rebuilt from static inputs, +// so the manager can restore the safe subset or refetch dynamic-only headers; +// v9 invalidated Kimi Code rows predating live effort and protocol metadata; +// v8 invalidated Codex discovery rows predating provider-native V2 compaction +// metadata; v7 invalidated rows predating the Antigravity Gemini budget-mode +// migration (cached specs still carrying `thinking.mode: "google-level"` and +// the old 3.5-flash effort routing); v6 invalidated rows that may contain the +// retired unknown-limit sentinels (222222/8888); v5 invalidated rows predating // effort-tier variant collapsing (raw `-low`/`-high`/`-thinking` member ids); // v4 dropped the pre-efforts ThinkingConfig shape. -const CACHE_SCHEMA_VERSION = 8; +const CACHE_SCHEMA_VERSION = 10; +const HEADER_RESTORE_VERSION = 1; interface CacheRow { provider_id: string; @@ -24,6 +30,9 @@ interface CacheRow { authoritative: number; static_fingerprint: string; models: string; + header_omitted_model_ids: string; + unrestorable_header_model_ids: string; + header_restore_version: number; } interface TableInfoRow { @@ -35,6 +44,12 @@ interface CacheEntry { fresh: boolean; authoritative: boolean; updatedAt: number; + /** Model ids whose live headers were intentionally omitted from disk. */ + headerOmittedModelIds: readonly string[]; + /** Header-bearing model ids that cannot be rebuilt from the static source. */ + unrestorableHeaderModelIds: readonly string[]; + /** Whether unrestorable markers predate request-model header matching. */ + legacyHeaderRestoreMarkers: boolean; /** * Hash of the static catalog slice that was merged into `models` when this * row was written. `resolveProviderModels` compares against the current @@ -52,6 +67,10 @@ function openDb(resolvedPath: string): Database { // Install the busy handler BEFORE any lock-taking statement. See // https://github.com/can1357/oh-my-pi/issues/2421. db.run("PRAGMA busy_timeout = 3000"); + // Schema invalidation can delete rows containing credentials written by old + // versions. Overwrite deleted SQLite cells instead of leaving their bytes in + // free pages where a raw scan of models.db can still recover them (#5780). + db.run("PRAGMA secure_delete = ON"); db.run("PRAGMA journal_mode = WAL"); db.run(` CREATE TABLE IF NOT EXISTS model_cache ( @@ -60,6 +79,9 @@ function openDb(resolvedPath: string): Database { updated_at INTEGER NOT NULL, authoritative INTEGER NOT NULL DEFAULT 0, static_fingerprint TEXT NOT NULL DEFAULT '', + header_omitted_model_ids TEXT NOT NULL DEFAULT '[]', + unrestorable_header_model_ids TEXT NOT NULL DEFAULT '[]', + header_restore_version INTEGER NOT NULL DEFAULT 0, models TEXT NOT NULL ) `); @@ -98,6 +120,18 @@ function migrateCacheSchema(db: Database): void { if (!columns.some(column => column.name === "static_fingerprint")) { db.run("ALTER TABLE model_cache ADD COLUMN static_fingerprint TEXT NOT NULL DEFAULT ''"); } + if (!columns.some(column => column.name === "header_omitted_model_ids")) { + db.run("ALTER TABLE model_cache ADD COLUMN header_omitted_model_ids TEXT NOT NULL DEFAULT '[]'"); + } + if (!columns.some(column => column.name === "unrestorable_header_model_ids")) { + db.run("ALTER TABLE model_cache ADD COLUMN unrestorable_header_model_ids TEXT NOT NULL DEFAULT '[]'"); + } + if (!columns.some(column => column.name === "header_restore_version")) { + // Existing v10 rows get 0, distinguishing markers produced by the + // old id-only header matcher from rows written after request-model + // header matching was introduced. + db.run("ALTER TABLE model_cache ADD COLUMN header_restore_version INTEGER NOT NULL DEFAULT 0"); + } } finally { stmt.finalize(); } @@ -124,6 +158,14 @@ export function readModelCache( return null; } const models = JSON.parse(row.models) as ModelSpec[]; + const parsedHeaderModelIds: unknown = JSON.parse(row.header_omitted_model_ids); + const headerOmittedModelIds = Array.isArray(parsedHeaderModelIds) + ? parsedHeaderModelIds.filter((id): id is string => typeof id === "string") + : []; + const parsedUnrestorableModelIds: unknown = JSON.parse(row.unrestorable_header_model_ids); + const unrestorableHeaderModelIds = Array.isArray(parsedUnrestorableModelIds) + ? parsedUnrestorableModelIds.filter((id): id is string => typeof id === "string") + : []; const ageMs = now() - row.updated_at; const fresh = Number.isFinite(ageMs) && ageMs >= 0 && ageMs <= ttlMs; return { @@ -131,6 +173,9 @@ export function readModelCache( fresh, authoritative: row.authoritative === 1, updatedAt: row.updated_at, + headerOmittedModelIds, + unrestorableHeaderModelIds, + legacyHeaderRestoreMarkers: row.header_restore_version < HEADER_RESTORE_VERSION, staticFingerprint: row.static_fingerprint ?? "", }; } finally { @@ -142,6 +187,39 @@ export function readModelCache( } } +/** Whether a live model carries at least one request header. */ +function hasModelHeaders(model: Model): boolean { + const headers = model.headers; + if (!headers) return false; + for (const _key in headers) return true; + return false; +} + +/** + * Project a live model to cache-safe metadata. + * + * Headers are never persisted: custom/runtime providers may use arbitrary + * credential header names, so no name-based filter can be complete. The + * separately persisted model-id list lets the manager restore matching static + * headers and reject/refetch dynamic-only cached models that need live headers. + */ +function toCachedModelSpec(model: Model): ModelSpec { + const { headers: _headers, compatConfig, ...rest } = model; + return { ...rest, compat: compatConfig }; +} + +/** Whether two in-memory header records are byte-for-byte equivalent. */ +function headersEqual(left: Record | undefined, right: Record | undefined): boolean { + if (!left || !right) return left === right; + for (const key in left) { + if (right[key] !== left[key]) return false; + } + for (const key in right) { + if (!(key in left)) return false; + } + return true; +} + export function writeModelCache( providerId: string, updatedAt: number, @@ -149,19 +227,45 @@ export function writeModelCache( authoritative: boolean, staticFingerprint: string, dbPath?: string, + staticHeaderSources: readonly Model[] = [], ): void { try { withModelCacheDb(dbPath, db => { + const headerOmittedModelIds: string[] = []; + const unrestorableHeaderModelIds: string[] = []; + const cachedModels: ModelSpec[] = []; + const staticById = new Map(staticHeaderSources.map(model => [model.id, model])); + for (const model of models) { + if (hasModelHeaders(model)) { + headerOmittedModelIds.push(model.id); + // Synthesized variants (e.g. Copilot `-1m`) have no same-id static + // entry; their headers come from the `requestModelId` base. Match + // against that source too, else they are wrongly flagged + // unrestorable and dropped on the next offline read (#6037, #6284). + const staticHeaderSource = + staticById.get(model.id) ?? (model.requestModelId ? staticById.get(model.requestModelId) : undefined); + if (!headersEqual(model.headers, staticHeaderSource?.headers)) { + unrestorableHeaderModelIds.push(model.id); + } + } + cachedModels.push(toCachedModelSpec(model)); + } db.run( - `INSERT OR REPLACE INTO model_cache (provider_id, version, updated_at, authoritative, static_fingerprint, models) - VALUES (?, ?, ?, ?, ?, ?)`, + `INSERT OR REPLACE INTO model_cache ( + provider_id, version, updated_at, authoritative, static_fingerprint, + header_omitted_model_ids, unrestorable_header_model_ids, + header_restore_version, models + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)`, [ providerId, CACHE_SCHEMA_VERSION, updatedAt, authoritative ? 1 : 0, staticFingerprint, - JSON.stringify(models.map(model => ({ ...model, compat: model.compatConfig, compatConfig: undefined }))), + JSON.stringify(headerOmittedModelIds), + JSON.stringify(unrestorableHeaderModelIds), + HEADER_RESTORE_VERSION, + JSON.stringify(cachedModels), ], ); }); diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index 074fbd2e2..1fa5a3b7c 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -100,6 +100,57 @@ function passModelList(value: unknown): Model[] { } return out; } +interface CachedHeaderRestoreResult { + models: Model[]; + unresolvedModelIds: ReadonlySet; +} + +/** + * Restore cache-omitted headers from the current static source. + * + * A same-id static match is trusted only when the row did not flag the model + * unrestorable (its live headers matched static when cached). Request-model + * fallback also honors that marker for current rows. Only legacy rows written + * before request-model header matching may bypass it: their id-only writer + * necessarily marked every synthesized variant unrestorable (#6037, #6284). + * Header-bearing models without a trusted source cannot be reconstructed + * safely without persisting arbitrary credential values; callers must refetch + * them online or omit them rather than return a broken model. + */ +function restoreCachedModelHeaders( + cachedModels: readonly ModelSpec[], + staticModels: readonly Model[], + headerOmittedModelIds: readonly string[], + unrestorableHeaderModelIds: readonly string[], + legacyHeaderRestoreMarkers: boolean, +): CachedHeaderRestoreResult { + const models = passModelList(cachedModels); + if (headerOmittedModelIds.length === 0) { + return { models, unresolvedModelIds: new Set() }; + } + const omittedIds = new Set(headerOmittedModelIds); + const unrestorableIds = new Set(unrestorableHeaderModelIds); + const staticById = new Map(staticModels.map(model => [model.id, model])); + const unresolvedModelIds = new Set(); + const restored = models.map(model => { + if (!omittedIds.has(model.id)) return model; + const unrestorable = unrestorableIds.has(model.id); + // Current unrestorable markers prove that neither same-id nor request-model + // static headers matched the live model. Only the old id-only writer's + // markers may recover a synthesized variant through `requestModelId`. + const staticModel = unrestorable + ? legacyHeaderRestoreMarkers && model.requestModelId + ? staticById.get(model.requestModelId) + : undefined + : (staticById.get(model.id) ?? (model.requestModelId ? staticById.get(model.requestModelId) : undefined)); + if (!staticModel?.headers) { + unresolvedModelIds.add(model.id); + return model; + } + return { ...model, headers: staticModel.headers }; + }); + return { models: restored, unresolvedModelIds }; +} /** * Resolves provider models with source precedence: @@ -119,10 +170,20 @@ export async function resolveProviderModels(options.staticModels) : (getBundledModels(options.providerId as GeneratedProvider) as Model[]); const cache = readModelCache(cacheProviderId, ttlMs, now, dbPath); + const restoredCache = restoreCachedModelHeaders( + cache?.models ?? [], + staticModels, + cache?.headerOmittedModelIds ?? [], + cache?.unrestorableHeaderModelIds ?? [], + cache?.legacyHeaderRestoreMarkers ?? false, + ); + const usableCachedModels = restoredCache.models.filter(model => !restoredCache.unresolvedModelIds.has(model.id)); + const cacheHasUnresolvedHeaders = restoredCache.unresolvedModelIds.size > 0; const dynamicModelsAuthoritative = options.dynamicModelsAuthoritative ?? false; const staticFingerprint = fingerprintStatic(staticModels, dynamicModelsAuthoritative); const cacheFingerprintMatches = cache?.staticFingerprint === staticFingerprint && staticFingerprint.length > 0; - const hasUsableFreshCache = (cache?.fresh ?? false) && (!dynamicModelsAuthoritative || cacheFingerprintMatches); + const hasUsableFreshCache = + (cache?.fresh ?? false) && !cacheHasUnresolvedHeaders && (!dynamicModelsAuthoritative || cacheFingerprintMatches); const dynamicFetcher = options.fetchDynamicModels; const hasDynamicFetcher = typeof dynamicFetcher === "function"; const hasAuthoritativeCache = ((cache?.authoritative ?? false) && hasUsableFreshCache) || !hasDynamicFetcher; @@ -139,8 +200,14 @@ export async function resolveProviderModels(cache.models)), stale: false }; + if ( + !shouldFetchFromNetwork && + cache?.fresh && + hasAuthoritativeCache && + cacheFingerprintMatches && + !cacheHasUnresolvedHeaders + ) { + return { models: collapseBuiltModelVariants(restoredCache.models), stale: false }; } const [fetchedModelsDevModels, fetchedDynamicModels] = shouldFetchFromNetwork @@ -153,7 +220,7 @@ export async function resolveProviderModels(cache?.models ?? []), + usableCachedModels, staticModels, cacheFingerprintMatches, options.dropCachedModelIdsOnStaticMismatch, @@ -178,11 +245,22 @@ export async function resolveProviderModels(cacheProviderId, ttlMs, now, dbPath); + const latestRestoredCache = restoreCachedModelHeaders( + latestCache?.models ?? cache?.models ?? [], + staticModels, + latestCache?.headerOmittedModelIds ?? cache?.headerOmittedModelIds ?? [], + latestCache?.unrestorableHeaderModelIds ?? cache?.unrestorableHeaderModelIds ?? [], + latestCache?.legacyHeaderRestoreMarkers ?? cache?.legacyHeaderRestoreMarkers ?? false, + ); + const latestUsableCacheModels = latestRestoredCache.models.filter( + model => !latestRestoredCache.unresolvedModelIds.has(model.id), + ); writeModelCache( cacheProviderId, now(), @@ -190,7 +268,7 @@ export async function resolveProviderModels(latestCache?.models ?? cache?.models ?? []), + latestUsableCacheModels, staticModels, cacheFingerprintMatches, options.dropCachedModelIdsOnStaticMismatch, @@ -200,6 +278,7 @@ export async function resolveProviderModels( (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") && supportsAdaptiveThinkingDisplay(spec.id); const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id); - if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort) { + const needsDefaultLevel = thinking.defaultLevel === undefined && isKimiK3ModelId(spec.id); + if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) { return thinking; } const filled: ThinkingConfig = { ...thinking }; @@ -191,6 +195,9 @@ function fillThinkingWireDefaults( if (needsDisplay) { filled.supportsDisplay = true; } + if (needsDefaultLevel) { + filled.defaultLevel = Effort.Max; + } if (needsRequiresEffort) { filled.requiresEffort = true; } @@ -208,6 +215,9 @@ export function deriveThinking(spec: ModelSpec, compat: mode: inferThinkingControlMode(spec, parsed), efforts, }; + if (isKimiK3ModelId(spec.id)) { + config.defaultLevel = Effort.Max; + } const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts); if (effortMap !== undefined) { config.effortMap = effortMap; @@ -328,6 +338,9 @@ function getModelDefinedEfforts( return DEFAULT_REASONING_EFFORTS_WITH_MAX; } } + if (isKimiK3ModelId(spec.id)) { + return KIMI_K3_REASONING_EFFORTS; + } if (isSakanaFuguReasoningModel(spec)) { return HIGH_MAX_REASONING_EFFORTS; } @@ -544,6 +557,7 @@ function impliesMandatoryReasoning(parsed: ParsedModel, modelId: string): boolea if (semverGte(parsed.version, "3.0")) return true; if (parsed.kind === "pro" && semverGte(parsed.version, "2.5")) return true; } + if (isKimiK3ModelId(modelId)) return true; if (isMinimaxM2FamilyModelId(modelId)) return true; if (OPENAI_O_SERIES_RE.test(bareModelId(modelId))) return true; return findThinkingVariantToken(modelId) !== undefined; diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index c7fdd1695..24eb18fe9 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -3036,11 +3036,11 @@ }, "glm-4.5": { "id": "glm-4.5", - "name": "glm-4.5", + "name": "GLM-4.5", "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -3051,7 +3051,17 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 98304 + "maxTokens": 98304, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "glm-4.5-air": { "id": "glm-4.5-air", @@ -3084,7 +3094,7 @@ }, "glm-4.6": { "id": "glm-4.6", - "name": "glm-4.6", + "name": "GLM-4.6", "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", @@ -8362,10 +8372,10 @@ "image" ], "cost": { - "input": 5, - "output": 25, - "cacheRead": 0.5, - "cacheWrite": 6.25 + "input": 5.5, + "output": 27.5, + "cacheRead": 0.55, + "cacheWrite": 6.875 }, "contextWindow": 200000, "maxTokens": 64000, @@ -8420,10 +8430,10 @@ "image" ], "cost": { - "input": 5, - "output": 25, - "cacheRead": 0.5, - "cacheWrite": 6.25 + "input": 5.5, + "output": 27.5, + "cacheRead": 0.55, + "cacheWrite": 6.875 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -8567,10 +8577,10 @@ "image" ], "cost": { - "input": 2, - "output": 10, - "cacheRead": 0.2, - "cacheWrite": 2.5 + "input": 2.2, + "output": 11, + "cacheRead": 0.22, + "cacheWrite": 2.75 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -8616,7 +8626,7 @@ }, "global.anthropic.claude-fable-5": { "id": "global.anthropic.claude-fable-5", - "name": "Claude Fable 5 (Global)", + "name": "Claude Fable 5", "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", @@ -8675,7 +8685,7 @@ }, "global.anthropic.claude-opus-4-5-20251101-v1:0": { "id": "global.anthropic.claude-opus-4-5-20251101-v1:0", - "name": "Claude Opus 4.5", + "name": "Claude Opus 4.5 (Global)", "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", @@ -8822,7 +8832,7 @@ }, "global.anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "global.anthropic.claude-sonnet-4-5-20250929-v1:0", - "name": "Claude Sonnet 4.5 (Global)", + "name": "Claude Sonnet 4.5", "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", @@ -8851,7 +8861,7 @@ }, "global.anthropic.claude-sonnet-4-6": { "id": "global.anthropic.claude-sonnet-4-6", - "name": "Claude Sonnet 4.6", + "name": "Claude Sonnet 4.6 (Global)", "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", @@ -9684,6 +9694,96 @@ }, "contextPromotionTarget": "amazon-bedrock/openai.gpt-5.4" }, + "openai.gpt-5.6-luna": { + "id": "openai.gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "openai.gpt-5.6-sol": { + "id": "openai.gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "openai.gpt-5.6-terra": { + "id": "openai.gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, "openai.gpt-oss-120b": { "id": "openai.gpt-oss-120b", "name": "gpt-oss-120b", @@ -10155,7 +10255,7 @@ }, "us.anthropic.claude-opus-4-1-20250805-v1:0": { "id": "us.anthropic.claude-opus-4-1-20250805-v1:0", - "name": "Claude Opus 4.1", + "name": "Claude Opus 4.1 (US)", "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", @@ -10448,7 +10548,7 @@ }, "us.deepseek.r1-v1:0": { "id": "us.deepseek.r1-v1:0", - "name": "DeepSeek-R1", + "name": "DeepSeek-R1 (US)", "api": "bedrock-converse-stream", "provider": "amazon-bedrock", "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", @@ -11732,7 +11832,7 @@ "cacheRead": 0.13, "cacheWrite": 0 }, - "contextWindow": 272000, + "contextWindow": 400000, "maxTokens": 128000, "thinking": { "mode": "effort", @@ -11790,7 +11890,7 @@ "cacheRead": 0.03, "cacheWrite": 0 }, - "contextWindow": 272000, + "contextWindow": 400000, "maxTokens": 128000, "thinking": { "mode": "effort", @@ -11819,7 +11919,7 @@ "cacheRead": 0.01, "cacheWrite": 0 }, - "contextWindow": 272000, + "contextWindow": 400000, "maxTokens": 128000, "thinking": { "mode": "effort", @@ -11877,7 +11977,7 @@ "cacheRead": 0.125, "cacheWrite": 0 }, - "contextWindow": 272000, + "contextWindow": 400000, "maxTokens": 128000, "thinking": { "mode": "effort", @@ -12294,6 +12394,126 @@ }, "contextPromotionTarget": "azure/gpt-5.4" }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "gpt-chat-latest": { + "id": "gpt-chat-latest", + "name": "GPT Chat Latest", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "o1": { "id": "o1", "name": "o1", @@ -12649,6 +12869,26 @@ ] } }, + "thinkingmachines/inkling": { + "id": "thinkingmachines/inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 4.05, + "cacheRead": 0.16999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 32768 + }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM 4.7", @@ -12749,11 +12989,11 @@ "cost": { "input": 1.4, "output": 4.4, - "cacheRead": 0.26, + "cacheRead": 0.14, "cacheWrite": 0 }, - "contextWindow": 202720, - "maxTokens": 202720, + "contextWindow": 256000, + "maxTokens": 256000, "thinking": { "mode": "effort", "efforts": [ @@ -12913,7 +13153,7 @@ "cost": { "input": 2.25, "output": 2.75, - "cacheRead": 0, + "cacheRead": 2.25, "cacheWrite": 0 }, "contextWindow": 131072, @@ -13442,6 +13682,36 @@ ] } }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "anthropic-messages", + "provider": "cloudflare-ai-gateway", + "baseUrl": "https://gateway.ai.cloudflare.com/v1///anthropic", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "requiresEffort": true + } + }, "openai/gpt-4": { "id": "openai/gpt-4", "name": "GPT-4", @@ -13725,6 +13995,96 @@ }, "contextPromotionTarget": "cloudflare-ai-gateway/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "anthropic-messages", + "provider": "cloudflare-ai-gateway", + "baseUrl": "https://gateway.ai.cloudflare.com/v1///anthropic", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "anthropic-messages", + "provider": "cloudflare-ai-gateway", + "baseUrl": "https://gateway.ai.cloudflare.com/v1///anthropic", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "anthropic-messages", + "provider": "cloudflare-ai-gateway", + "baseUrl": "https://gateway.ai.cloudflare.com/v1///anthropic", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, "openai/o1": { "id": "openai/o1", "name": "o1", @@ -13996,6 +14356,35 @@ "xhigh" ] } + }, + "workers-ai/@cf/zai-org/glm-5.2": { + "id": "workers-ai/@cf/zai-org/glm-5.2", + "name": "Glm 5.2", + "api": "anthropic-messages", + "provider": "cloudflare-ai-gateway", + "baseUrl": "https://gateway.ai.cloudflare.com/v1///anthropic", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } } }, "coreweave": { @@ -14262,6 +14651,36 @@ "requiresEffort": true } }, + "MiniMaxAI/MiniMax-M3": { + "id": "MiniMaxAI/MiniMax-M3", + "name": "MiniMax M3", + "api": "openai-completions", + "provider": "coreweave", + "baseUrl": "https://api.inference.wandb.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.29, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", @@ -15829,114 +16248,9 @@ } }, "devin": { - "claude-5-fable-high": { - "id": "claude-5-fable-high", - "name": "Claude Fable 5 High", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-5-fable-low": { - "id": "claude-5-fable-low", - "name": "Claude Fable 5 Low", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-5-fable-max": { - "id": "claude-5-fable-max", - "name": "Claude Fable 5 Max", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-5-fable-medium": { - "id": "claude-5-fable-medium", - "name": "Claude Fable 5 Medium", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-5-fable-xhigh": { - "id": "claude-5-fable-xhigh", - "name": "Claude Fable 5 XHigh", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-opus-4-6": { - "id": "claude-opus-4-6", - "name": "Claude Opus 4.6", + "swe-1-6-slow": { + "id": "swe-1-6-slow", + "name": "SWE-1.6 Slow", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -15953,1540 +16267,6 @@ "cacheWrite": 0 }, "contextWindow": 200000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortRouting": { - "off": "claude-opus-4-6", - "minimal": "claude-opus-4-6-thinking", - "low": "claude-opus-4-6-thinking", - "medium": "claude-opus-4-6-thinking", - "high": "claude-opus-4-6-thinking" - } - } - }, - "claude-opus-4-6-1m": { - "id": "claude-opus-4-6-1m", - "name": "Claude Opus 4.6 1M", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortRouting": { - "off": "claude-opus-4-6-1m", - "minimal": "claude-opus-4-6-thinking-1m", - "low": "claude-opus-4-6-thinking-1m", - "medium": "claude-opus-4-6-thinking-1m", - "high": "claude-opus-4-6-thinking-1m" - } - } - }, - "claude-opus-4-7": { - "id": "claude-opus-4-7", - "name": "Claude Opus 4.7", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "requiresEffort": true, - "effortRouting": { - "low": "claude-opus-4-7-low", - "medium": "claude-opus-4-7-medium", - "high": "claude-opus-4-7-high", - "xhigh": "claude-opus-4-7-xhigh", - "max": "claude-opus-4-7-max" - } - }, - "requestModelId": "claude-opus-4-7-low" - }, - "claude-opus-4-7-fast": { - "id": "claude-opus-4-7-fast", - "name": "Claude Opus 4.7 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "requiresEffort": true, - "effortRouting": { - "low": "claude-opus-4-7-low-fast", - "medium": "claude-opus-4-7-medium-fast", - "high": "claude-opus-4-7-high-fast", - "xhigh": "claude-opus-4-7-xhigh-fast", - "max": "claude-opus-4-7-max-fast" - } - }, - "requestModelId": "claude-opus-4-7-low-fast" - }, - "claude-opus-4-8": { - "id": "claude-opus-4-8", - "name": "Claude Opus 4.8", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "requiresEffort": true, - "effortRouting": { - "low": "claude-opus-4-8-low", - "medium": "claude-opus-4-8-medium", - "high": "claude-opus-4-8-high", - "xhigh": "claude-opus-4-8-xhigh", - "max": "claude-opus-4-8-max" - } - }, - "requestModelId": "claude-opus-4-8-low" - }, - "claude-opus-4-8-fast": { - "id": "claude-opus-4-8-fast", - "name": "Claude Opus 4.8 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "requiresEffort": true, - "effortRouting": { - "low": "claude-opus-4-8-low-fast", - "medium": "claude-opus-4-8-medium-fast", - "high": "claude-opus-4-8-high-fast", - "xhigh": "claude-opus-4-8-xhigh-fast", - "max": "claude-opus-4-8-max-fast" - } - }, - "requestModelId": "claude-opus-4-8-low-fast" - }, - "claude-sonnet-4-6": { - "id": "claude-sonnet-4-6", - "name": "Claude Sonnet 4.6", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortRouting": { - "off": "claude-sonnet-4-6", - "minimal": "claude-sonnet-4-6-thinking", - "low": "claude-sonnet-4-6-thinking", - "medium": "claude-sonnet-4-6-thinking", - "high": "claude-sonnet-4-6-thinking" - } - } - }, - "claude-sonnet-4-6-1m": { - "id": "claude-sonnet-4-6-1m", - "name": "Claude Sonnet 4.6 1M", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortRouting": { - "off": "claude-sonnet-4-6-1m", - "minimal": "claude-sonnet-4-6-thinking-1m", - "low": "claude-sonnet-4-6-thinking-1m", - "medium": "claude-sonnet-4-6-thinking-1m", - "high": "claude-sonnet-4-6-thinking-1m" - } - } - }, - "claude-sonnet-5-high": { - "id": "claude-sonnet-5-high", - "name": "Claude Sonnet 5 High", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-sonnet-5-low": { - "id": "claude-sonnet-5-low", - "name": "Claude Sonnet 5 Low", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-sonnet-5-max": { - "id": "claude-sonnet-5-max", - "name": "Claude Sonnet 5 Max", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-sonnet-5-medium": { - "id": "claude-sonnet-5-medium", - "name": "Claude Sonnet 5 Medium", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "claude-sonnet-5-xhigh": { - "id": "claude-sonnet-5-xhigh", - "name": "Claude Sonnet 5 XHigh", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "deepseek-v4": { - "id": "deepseek-v4", - "name": "DeepSeek V4 Pro", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 64000 - }, - "gemini-3-1-pro": { - "id": "gemini-3-1-pro", - "name": "Gemini 3.1 Pro", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "high" - ], - "requiresEffort": true, - "effortRouting": { - "low": "gemini-3-1-pro-low", - "high": "gemini-3-1-pro-high" - } - }, - "requestModelId": "gemini-3-1-pro-low" - }, - "gemini-3-5-flash": { - "id": "gemini-3-5-flash", - "name": "Gemini 3.5 Flash", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true, - "effortRouting": { - "minimal": "gemini-3-5-flash-minimal", - "low": "gemini-3-5-flash-low", - "medium": "gemini-3-5-flash-medium", - "high": "gemini-3-5-flash-high" - } - }, - "requestModelId": "gemini-3-5-flash-minimal" - }, - "gemini-3-flash": { - "id": "gemini-3-flash", - "name": "Gemini 3 Flash", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true, - "effortRouting": { - "minimal": "MODEL_GOOGLE_GEMINI_3_0_FLASH_MINIMAL", - "low": "MODEL_GOOGLE_GEMINI_3_0_FLASH_LOW", - "medium": "MODEL_GOOGLE_GEMINI_3_0_FLASH_MEDIUM", - "high": "MODEL_GOOGLE_GEMINI_3_0_FLASH_HIGH" - } - }, - "requestModelId": "MODEL_GOOGLE_GEMINI_3_0_FLASH_MINIMAL" - }, - "glm-5-2": { - "id": "glm-5-2", - "name": "GLM-5.2 High", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "glm-5-2-1m": { - "id": "glm-5-2-1m", - "name": "GLM-5.2 High 1M", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "glm-5-2-max": { - "id": "glm-5-2-max", - "name": "GLM-5.2 Max", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "glm-5-2-max-1m": { - "id": "glm-5-2-max-1m", - "name": "GLM-5.2 Max 1M", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "glm-5-2-none": { - "id": "glm-5-2-none", - "name": "GLM-5.2 No Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "glm-5-2-none-1m": { - "id": "glm-5-2-none-1m", - "name": "GLM-5.2 No Thinking 1M", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-2": { - "id": "gpt-5-2", - "name": "GPT-5.2", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 384000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "MODEL_GPT_5_2_NONE", - "low": "MODEL_GPT_5_2_LOW", - "medium": "MODEL_GPT_5_2_MEDIUM", - "high": "MODEL_GPT_5_2_HIGH", - "xhigh": "MODEL_GPT_5_2_XHIGH" - } - }, - "requestModelId": "MODEL_GPT_5_2_NONE" - }, - "gpt-5-3-codex": { - "id": "gpt-5-3-codex", - "name": "GPT-5.3 Codex", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 400000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "requiresEffort": true, - "effortRouting": { - "low": "gpt-5-3-codex-low", - "medium": "gpt-5-3-codex-medium", - "high": "gpt-5-3-codex-high", - "xhigh": "gpt-5-3-codex-xhigh" - } - }, - "requestModelId": "gpt-5-3-codex-low" - }, - "gpt-5-3-codex-fast": { - "id": "gpt-5-3-codex-fast", - "name": "GPT-5.3 Codex Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 400000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "requiresEffort": true, - "effortRouting": { - "low": "gpt-5-3-codex-low-priority", - "medium": "gpt-5-3-codex-medium-priority", - "high": "gpt-5-3-codex-high-priority", - "xhigh": "gpt-5-3-codex-xhigh-priority" - } - }, - "requestModelId": "gpt-5-3-codex-low-priority" - }, - "gpt-5-4": { - "id": "gpt-5-4", - "name": "GPT-5.4", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-4-none", - "low": "gpt-5-4-low", - "medium": "gpt-5-4-medium", - "high": "gpt-5-4-high", - "xhigh": "gpt-5-4-xhigh" - } - }, - "requestModelId": "gpt-5-4-none" - }, - "gpt-5-4-fast": { - "id": "gpt-5-4-fast", - "name": "GPT-5.4 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-4-none-priority", - "low": "gpt-5-4-low-priority", - "medium": "gpt-5-4-medium-priority", - "high": "gpt-5-4-high-priority", - "xhigh": "gpt-5-4-xhigh-priority" - } - }, - "requestModelId": "gpt-5-4-none-priority" - }, - "gpt-5-4-mini": { - "id": "gpt-5-4-mini", - "name": "GPT-5.4 Mini", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 400000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "requiresEffort": true, - "effortRouting": { - "low": "gpt-5-4-mini-low", - "medium": "gpt-5-4-mini-medium", - "high": "gpt-5-4-mini-high", - "xhigh": "gpt-5-4-mini-xhigh" - } - }, - "requestModelId": "gpt-5-4-mini-low" - }, - "gpt-5-5": { - "id": "gpt-5-5", - "name": "GPT-5.5", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-5-none", - "low": "gpt-5-5-low", - "medium": "gpt-5-5-medium", - "high": "gpt-5-5-high", - "xhigh": "gpt-5-5-xhigh" - } - }, - "requestModelId": "gpt-5-5-none" - }, - "gpt-5-5-fast": { - "id": "gpt-5-5-fast", - "name": "GPT-5.5 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-5-none-priority", - "low": "gpt-5-5-low-priority", - "medium": "gpt-5-5-medium-priority", - "high": "gpt-5-5-high-priority", - "xhigh": "gpt-5-5-xhigh-priority" - } - }, - "requestModelId": "gpt-5-5-none-priority" - }, - "gpt-5-6-luna": { - "id": "gpt-5-6-luna", - "name": "GPT-5.6 Luna", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "effortRouting": { - "off": "gpt-5-6-luna-none", - "low": "gpt-5-6-luna-low", - "medium": "gpt-5-6-luna-medium", - "high": "gpt-5-6-luna-high", - "xhigh": "gpt-5-6-luna-xhigh", - "max": "gpt-5-6-luna-max" - } - }, - "requestModelId": "gpt-5-6-luna-none" - }, - "gpt-5-6-luna-fast": { - "id": "gpt-5-6-luna-fast", - "name": "GPT-5.6 Luna Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-6-luna-none-priority", - "low": "gpt-5-6-luna-low-priority", - "medium": "gpt-5-6-luna-medium-priority", - "high": "gpt-5-6-luna-high-priority", - "xhigh": "gpt-5-6-luna-xhigh-priority" - } - }, - "requestModelId": "gpt-5-6-luna-none-priority" - }, - "gpt-5-6-sol": { - "id": "gpt-5-6-sol", - "name": "GPT-5.6 Sol", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "effortRouting": { - "off": "gpt-5-6-sol-none", - "low": "gpt-5-6-sol-low", - "medium": "gpt-5-6-sol-medium", - "high": "gpt-5-6-sol-high", - "xhigh": "gpt-5-6-sol-xhigh", - "max": "gpt-5-6-sol-max" - } - }, - "requestModelId": "gpt-5-6-sol-none" - }, - "gpt-5-6-sol-fast": { - "id": "gpt-5-6-sol-fast", - "name": "GPT-5.6 Sol Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-6-sol-none-priority", - "low": "gpt-5-6-sol-low-priority", - "medium": "gpt-5-6-sol-medium-priority", - "high": "gpt-5-6-sol-high-priority", - "xhigh": "gpt-5-6-sol-xhigh-priority" - } - }, - "requestModelId": "gpt-5-6-sol-none-priority" - }, - "gpt-5-6-terra": { - "id": "gpt-5-6-terra", - "name": "GPT-5.6 Terra", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ], - "effortRouting": { - "off": "gpt-5-6-terra-none", - "low": "gpt-5-6-terra-low", - "medium": "gpt-5-6-terra-medium", - "high": "gpt-5-6-terra-high", - "xhigh": "gpt-5-6-terra-xhigh", - "max": "gpt-5-6-terra-max" - } - }, - "requestModelId": "gpt-5-6-terra-none" - }, - "gpt-5-6-terra-fast": { - "id": "gpt-5-6-terra-fast", - "name": "GPT-5.6 Terra Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ], - "effortRouting": { - "off": "gpt-5-6-terra-none-priority", - "low": "gpt-5-6-terra-low-priority", - "medium": "gpt-5-6-terra-medium-priority", - "high": "gpt-5-6-terra-high-priority", - "xhigh": "gpt-5-6-terra-xhigh-priority" - } - }, - "requestModelId": "gpt-5-6-terra-none-priority" - }, - "grok-4-5-high": { - "id": "grok-4-5-high", - "name": "Grok 4.5 High", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 500000, - "maxTokens": 64000 - }, - "grok-4-5-low": { - "id": "grok-4-5-low", - "name": "Grok 4.5 Low", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 500000, - "maxTokens": 64000 - }, - "grok-4-5-medium": { - "id": "grok-4-5-medium", - "name": "Grok 4.5 Medium", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 500000, - "maxTokens": 64000 - }, - "kimi-k2-6": { - "id": "kimi-k2-6", - "name": "Kimi K2.6", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 64000 - }, - "kimi-k2-7": { - "id": "kimi-k2-7", - "name": "Kimi K2.7", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 64000 - }, - "MODEL_CLAUDE_4_5_OPUS": { - "id": "MODEL_CLAUDE_4_5_OPUS", - "name": "Claude Opus 4.5", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "MODEL_CLAUDE_4_5_OPUS_THINKING": { - "id": "MODEL_CLAUDE_4_5_OPUS_THINKING", - "name": "Claude Opus 4.5 Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "MODEL_PRIVATE_11": { - "id": "MODEL_PRIVATE_11", - "name": "Claude Haiku 4.5", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "MODEL_PRIVATE_2": { - "id": "MODEL_PRIVATE_2", - "name": "Claude Sonnet 4.5", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "MODEL_PRIVATE_3": { - "id": "MODEL_PRIVATE_3", - "name": "Claude Sonnet 4.5 Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "MODEL_SWE_1_5": { - "id": "MODEL_SWE_1_5", - "name": "SWE-1.5 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 64000 - }, - "MODEL_SWE_1_5_SLOW": { - "id": "MODEL_SWE_1_5_SLOW", - "name": "SWE-1.5", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "nemotron-3-ultra-nvfp4": { - "id": "nemotron-3-ultra-nvfp4", - "name": "Nemotron 3 Ultra", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 64000 - }, - "swe-1-6": { - "id": "swe-1-6", - "name": "SWE-1.6", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "swe-1-6-fast": { - "id": "swe-1-6-fast", - "name": "SWE-1.6 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, - "swe-1-7": { - "id": "swe-1-7", - "name": "SWE-1.7", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262000, - "maxTokens": 64000 - }, - "swe-1-7-lightning": { - "id": "swe-1-7-lightning", - "name": "SWE-1.7 Lightning", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 202752, "maxTokens": 64000 } }, @@ -17723,9 +16503,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -17908,7 +16688,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -17941,7 +16721,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -19242,7 +18022,7 @@ "input": 1, "output": 6, "cacheRead": 0.1, - "cacheWrite": 0 + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -19276,7 +18056,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -19310,7 +18090,7 @@ "input": 2.5, "output": 15, "cacheRead": 0.25, - "cacheWrite": 0 + "cacheWrite": 3.125 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -19406,7 +18186,7 @@ "mai-code-1-flash-picker": { "id": "mai-code-1-flash-picker", "name": "MAI-Code-1-Flash", - "api": "openai-completions", + "api": "openai-responses", "provider": "github-copilot", "baseUrl": "https://api.githubcopilot.com", "reasoning": true, @@ -19425,18 +18205,14 @@ "User-Agent": "opencode/1.3.15", "X-GitHub-Api-Version": "2026-06-01" }, - "compat": { - "supportsStore": false, - "supportsDeveloperRole": false, - "supportsReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ "minimal", "low", "medium", - "high" + "high", + "xhigh" ] } }, @@ -20572,6 +19348,66 @@ "requiresEffort": true } }, + "gemini-3.5-flash-lite": { + "id": "gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "google-generative-ai", + "provider": "google", + "baseUrl": "https://generativelanguage.googleapis.com/v1beta", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "gemini-3.6-flash": { + "id": "gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "google-generative-ai", + "provider": "google", + "baseUrl": "https://generativelanguage.googleapis.com/v1beta", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "gemini-flash-latest": { "id": "gemini-flash-latest", "name": "Gemini Flash Latest", @@ -20584,9 +19420,9 @@ "image" ], "cost": { - "input": 0.3, - "output": 2.5, - "cacheRead": 0.075, + "input": 1.5, + "output": 9, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -20613,8 +19449,8 @@ "image" ], "cost": { - "input": 0.1, - "output": 0.4, + "input": 0.25, + "output": 1.5, "cacheRead": 0.025, "cacheWrite": 0 }, @@ -21349,6 +20185,43 @@ "suppressWhenOff": true } }, + "gemini-3.6-flash": { + "id": "gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "google-gemini-cli", + "provider": "google-antigravity", + "baseUrl": "https://daily-cloudcode-pa.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "requestModelId": "gemini-3.6-flash-low", + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true, + "effortRouting": { + "minimal": "gemini-3.6-flash-low", + "low": "gemini-3.6-flash-low", + "medium": "gemini-3.6-flash-medium", + "high": "gemini-3.6-flash-high" + } + } + }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", @@ -22148,6 +21021,66 @@ "requiresEffort": true } }, + "gemini-3.5-flash-lite": { + "id": "gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "gemini-3.6-flash": { + "id": "gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "gemini-flash-latest": { "id": "gemini-flash-latest", "name": "Gemini Flash Latest", @@ -22160,10 +21093,10 @@ "image" ], "cost": { - "input": 0.3, - "output": 2.5, - "cacheRead": 0.075, - "cacheWrite": 0.383 + "input": 1.5, + "output": 9, + "cacheRead": 0.15, + "cacheWrite": 0 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -22189,8 +21122,8 @@ "image" ], "cost": { - "input": 0.1, - "output": 0.4, + "input": 0.25, + "output": 1.5, "cacheRead": 0.025, "cacheWrite": 0 }, @@ -23908,6 +22841,33 @@ ] } }, + "XiaomiMiMo/MiMo-V2.5": { + "id": "XiaomiMiMo/MiMo-V2.5", + "name": "MiMo-V2.5", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.4, + "output": 2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, "XiaomiMiMo/MiMo-V2.5-Pro": { "id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "MiMo-V2.5-Pro", @@ -24320,9 +23280,9 @@ "image" ], "cost": { - "input": 0.5, - "output": 3, - "cacheRead": 0.05, + "input": 1.5, + "output": 9, + "cacheRead": 0.15, "cacheWrite": 0.08333333333333334 }, "contextWindow": 1048576, @@ -24458,6 +23418,25 @@ ] } }, + "~x-ai/grok-latest": { + "id": "~x-ai/grok-latest", + "name": "Grok Latest", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "ai21/jamba-large-1.7": { "id": "ai21/jamba-large-1.7", "name": "Jamba Large 1.7", @@ -26411,7 +25390,7 @@ }, "deepseek/deepseek-v4-flash:discounted": { "id": "deepseek/deepseek-v4-flash:discounted", - "name": "DeepSeek V4 Flash", + "name": "DeepSeek V4 Flash (lowest price)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -26475,7 +25454,7 @@ }, "deepseek/deepseek-v4-pro:discounted": { "id": "deepseek/deepseek-v4-pro:discounted", - "name": "DeepSeek V4 Pro", + "name": "DeepSeek V4 Pro (lowest price)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -27111,6 +26090,44 @@ "requiresEffort": true } }, + "google/gemini-3.5-flash-lite": { + "id": "google/gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536 + }, + "google/gemini-3.6-flash": { + "id": "google/gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536 + }, "google/gemma-2-27b-it": { "id": "google/gemma-2-27b-it", "name": "Gemma 2 27B", @@ -27805,6 +26822,25 @@ "contextWindow": null, "maxTokens": null }, + "kwaipilot/kat-coder-air-v2.5": { + "id": "kwaipilot/kat-coder-air-v2.5", + "name": "KAT-Coder-Air V2.5", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "kwaipilot/kat-coder-pro": { "id": "kwaipilot/kat-coder-pro", "name": "KAT-Coder-Pro V1", @@ -27843,6 +26879,44 @@ "contextWindow": 256000, "maxTokens": 80000 }, + "kwaipilot/kat-coder-pro-v2.5": { + "id": "kwaipilot/kat-coder-pro-v2.5", + "name": "KAT-Coder-Pro V2.5", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, + "kwaipilot/kat-coder-pro-v2.5:free": { + "id": "kwaipilot/kat-coder-pro-v2.5:free", + "name": "KAT-Coder-Pro V2.5 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "liquid/lfm-2-24b-a2b": { "id": "liquid/lfm-2-24b-a2b", "name": "LFM2-24B-A2B", @@ -27919,6 +26993,25 @@ "contextWindow": null, "maxTokens": null }, + "meituan/longcat-2.0": { + "id": "meituan/longcat-2.0", + "name": "LongCat 2.0", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "meituan/longcat-flash-chat": { "id": "meituan/longcat-flash-chat", "name": "LongCat Flash Chat", @@ -28244,6 +27337,25 @@ "contextWindow": null, "maxTokens": null }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "Muse Spark 1.1", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576 + }, "microsoft/phi-4": { "id": "microsoft/phi-4", "name": "Phi 4", @@ -29325,6 +28437,39 @@ ] } }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "morph-warp-grep-v2": { "id": "morph-warp-grep-v2", "name": "WarpGrep V2", @@ -31042,9 +30187,10 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -31053,7 +30199,17 @@ "cacheWrite": 0 }, "contextWindow": 372000, - "maxTokens": 128000 + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } }, "openai/gpt-5.6-luna-pro": { "id": "openai/gpt-5.6-luna-pro", @@ -31076,13 +30232,14 @@ }, "openai/gpt-5.6-sol": { "id": "openai/gpt-5.6-sol", - "name": "GPT-5.6 Sol (new)", + "name": "GPT-5.6 Sol", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -31091,7 +30248,17 @@ "cacheWrite": 0 }, "contextWindow": 372000, - "maxTokens": 128000 + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } }, "openai/gpt-5.6-sol-pro": { "id": "openai/gpt-5.6-sol-pro", @@ -31114,13 +30281,14 @@ }, "openai/gpt-5.6-terra": { "id": "openai/gpt-5.6-terra", - "name": "GPT-5.6 Terra (new)", + "name": "GPT-5.6 Terra", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -31129,7 +30297,17 @@ "cacheWrite": 0 }, "contextWindow": 372000, - "maxTokens": 128000 + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } }, "openai/gpt-5.6-terra-pro": { "id": "openai/gpt-5.6-terra-pro", @@ -31630,9 +30808,28 @@ ] } }, + "openrouter/auto-beta": { + "id": "openrouter/auto-beta", + "name": "OpenRouter Auto Router (Beta)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 2000000, + "maxTokens": null + }, "openrouter/bodybuilder": { "id": "openrouter/bodybuilder", - "name": "Body Builder (beta)", + "name": "OpenRouter Body Builder (beta)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -31786,7 +30983,7 @@ }, "openrouter/pareto-code": { "id": "openrouter/pareto-code", - "name": "Pareto Code Router", + "name": "OpenRouter Pareto Code Router", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -31974,6 +31171,44 @@ ] } }, + "poolside/laguna-s-2.1": { + "id": "poolside/laguna-s-2.1", + "name": "Laguna S 2.1", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072 + }, + "poolside/laguna-s-2.1:free": { + "id": "poolside/laguna-s-2.1:free", + "name": "Laguna S 2.1 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072 + }, "poolside/laguna-xs-2.1": { "id": "poolside/laguna-xs-2.1", "name": "Laguna XS 2.1", @@ -33827,6 +33062,25 @@ ] } }, + "stealth/gpt-5.6-sol": { + "id": "stealth/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, "stealth/qwen3.6-plus": { "id": "stealth/qwen3.6-plus", "name": "Qwen3.6 Plus", @@ -34143,6 +33397,25 @@ "contextWindow": 32768, "maxTokens": 32768 }, + "thinkingmachines/inkling": { + "id": "thinkingmachines/inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536 + }, "tngtech/deepseek-r1t2-chimera": { "id": "tngtech/deepseek-r1t2-chimera", "name": "DeepSeek R1T2 Chimera", @@ -34523,6 +33796,36 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", @@ -35167,9 +34470,43 @@ } }, "kimi-code": { + "k3": { + "id": "k3", + "name": "K3", + "api": "openai-completions", + "provider": "kimi-code", + "baseUrl": "https://api.kimi.com/coding/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 32000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + }, + "compat": { + "thinkingFormat": "kimi", + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + } + }, "kimi-for-coding": { "id": "kimi-for-coding", - "name": "K2.7 Code", + "name": "K2.7 Coding", "api": "openai-completions", "provider": "kimi-code", "baseUrl": "https://api.kimi.com/coding/v1", @@ -35198,6 +34535,45 @@ "medium", "high" ] + }, + "compat": { + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + } + }, + "kimi-for-coding-highspeed": { + "id": "kimi-for-coding-highspeed", + "name": "K2.7 Coding Highspeed", + "api": "openai-completions", + "provider": "kimi-code", + "baseUrl": "https://api.kimi.com/coding/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + }, + "compat": { + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false } }, "kimi-k2": { @@ -37287,6 +36663,39 @@ ] } }, + "kimi-k3": { + "id": "kimi-k3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "moonshot", + "baseUrl": "https://api.moonshot.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "moonshot-v1-128k": { "id": "moonshot-v1-128k", "name": "moonshot-v1-128k", @@ -38871,6 +38280,36 @@ "contextWindow": 256000, "maxTokens": null }, + "bytedance/doubao-seed-character": { + "id": "bytedance/doubao-seed-character", + "name": "Doubao-Seed-Character", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.1179, + "output": 0.2947, + "cacheRead": 0.0236, + "cacheWrite": 0.0025 + }, + "contextWindow": 128000, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "chutesai/Mistral-Small-3.2-24B-Instruct-2506": { "id": "chutesai/Mistral-Small-3.2-24B-Instruct-2506", "name": "chutesai/Mistral-Small-3.2-24B-Instruct-2506", @@ -43020,6 +42459,44 @@ "requiresEffort": true } }, + "google/gemini-3.5-flash-lite": { + "id": "google/gemini-3.5-flash-lite", + "name": "google/gemini-3.5-flash-lite", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536 + }, + "google/gemini-3.6-flash": { + "id": "google/gemini-3.6-flash", + "name": "google/gemini-3.6-flash", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536 + }, "google/gemini-flash-1.5": { "id": "google/gemini-flash-1.5", "name": "google/gemini-flash-1.5", @@ -43041,22 +42518,33 @@ }, "google/gemini-flash-latest": { "id": "google/gemini-flash-latest", - "name": "google/gemini-flash-latest", + "name": "Gemini Flash Latest", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.5, + "output": 9, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 65536 + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "google/gemini-flash-lite-latest": { "id": "google/gemini-flash-lite-latest", @@ -43926,6 +43414,25 @@ "contextWindow": null, "maxTokens": null }, + "kwaipilot/kat-coder-air-v2.5": { + "id": "kwaipilot/kat-coder-air-v2.5", + "name": "kwaipilot/kat-coder-air-v2.5", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "kwaipilot/kat-coder-pro-v2": { "id": "kwaipilot/kat-coder-pro-v2", "name": "KAT-Coder-Pro V2", @@ -43945,6 +43452,25 @@ "contextWindow": 256000, "maxTokens": 80000 }, + "kwaipilot/kat-coder-pro-v2.5": { + "id": "kwaipilot/kat-coder-pro-v2.5", + "name": "kwaipilot/kat-coder-pro-v2.5", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "LatitudeGames/Wayfarer-Large-70B-Llama-3.3": { "id": "LatitudeGames/Wayfarer-Large-70B-Llama-3.3", "name": "LatitudeGames/Wayfarer-Large-70B-Llama-3.3", @@ -44920,7 +44446,7 @@ "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -44931,7 +44457,17 @@ "cacheWrite": 0 }, "contextWindow": null, - "maxTokens": null + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "Magistral-Small-2506": { "id": "Magistral-Small-2506", @@ -46518,6 +46054,39 @@ "contextWindow": 262144, "maxTokens": 262144 }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "moonshotai/kimi-latest": { "id": "moonshotai/kimi-latest", "name": "Kimi Latest", @@ -46548,6 +46117,25 @@ ] } }, + "nano-gpt-help": { + "id": "nano-gpt-help", + "name": "nano-gpt-help", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "nanogpt/coding-router": { "id": "nanogpt/coding-router", "name": "Coding Router", @@ -48868,6 +48456,35 @@ ] } }, + "poolside/laguna-s-2.1": { + "id": "poolside/laguna-s-2.1", + "name": "poolside/laguna-s-2.1", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", "name": "Laguna XS.2", @@ -50563,6 +50180,25 @@ ] } }, + "qwen3.8-max-preview": { + "id": "qwen3.8-max-preview", + "name": "qwen3.8-max-preview", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "qwq-32b": { "id": "qwq-32b", "name": "qwq-32b", @@ -52432,6 +52068,36 @@ "contextWindow": null, "maxTokens": null }, + "thinkingmachines/inkling": { + "id": "thinkingmachines/inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 4.05, + "cacheRead": 0.16999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "THUDM/GLM-4-32B-0414": { "id": "THUDM/GLM-4-32B-0414", "name": "THUDM/GLM-4-32B-0414", @@ -53109,13 +52775,14 @@ }, "x-ai/grok-4.5": { "id": "x-ai/grok-4.5", - "name": "x-ai/grok-4.5", + "name": "Grok 4.5", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -53124,7 +52791,17 @@ "cacheWrite": 0 }, "contextWindow": 500000, - "maxTokens": 500000 + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", @@ -55609,6 +55286,40 @@ ] } }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "nousresearch/hermes-2-pro-llama-3-8b": { "id": "nousresearch/hermes-2-pro-llama-3-8b", "name": "Hermes 2 Pro Llama 3 8B", @@ -56481,9 +56192,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.14, + "output": 0.58, + "cacheRead": 0.035, "cacheWrite": 0 }, "contextWindow": 262144, @@ -60473,7 +60184,7 @@ }, "deepseek-v3.2": { "id": "deepseek-v3.2", - "name": "deepseek-v3.2", + "name": "DeepSeek V3.2", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -60603,7 +60314,7 @@ }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", - "name": "gemini-3-flash-preview", + "name": "Gemini 3 Flash Preview", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -60727,7 +60438,7 @@ }, "glm-4.6": { "id": "glm-4.6", - "name": "glm-4.6", + "name": "GLM-4.6", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -60756,7 +60467,7 @@ }, "glm-4.7": { "id": "glm-4.7", - "name": "glm-4.7", + "name": "GLM-4.7", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -60785,7 +60496,7 @@ }, "glm-5": { "id": "glm-5", - "name": "glm-5", + "name": "GLM-5", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -60928,7 +60639,7 @@ }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", - "name": "kimi-k2-thinking", + "name": "Kimi K2 Thinking", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -61389,7 +61100,7 @@ }, "qwen3-coder-next": { "id": "qwen3-coder-next", - "name": "qwen3-coder-next", + "name": "Qwen3 Coder Next", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", @@ -62992,361 +62703,6 @@ } }, "openai-codex": { - "codex-auto-review": { - "id": "codex-auto-review", - "name": "Codex Auto Review", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "remoteCompaction": { - "enabled": true, - "api": "openai-codex-responses", - "v2StreamingEnabled": true - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 43, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "gpt-5": { - "id": "gpt-5", - "name": "GPT-5", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.25, - "output": 10, - "cacheRead": 0.125, - "cacheWrite": 0 - }, - "contextWindow": 400000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 16, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5-codex": { - "id": "gpt-5-codex", - "name": "GPT-5-Codex", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.25, - "output": 10, - "cacheRead": 0.125, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 15, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5-codex-mini": { - "id": "gpt-5-codex-mini", - "name": "gpt-5-codex-mini", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 20, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.1": { - "id": "gpt-5.1", - "name": "GPT-5.1", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.25, - "output": 10, - "cacheRead": 0.13, - "cacheWrite": 0 - }, - "contextWindow": 400000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 12, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.1-codex": { - "id": "gpt-5.1-codex", - "name": "GPT-5.1 Codex", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.25, - "output": 10, - "cacheRead": 0.125, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 11, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.1-codex-max": { - "id": "gpt-5.1-codex-max", - "name": "GPT-5.1 Codex Max", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.25, - "output": 10, - "cacheRead": 0.125, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 10, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.1-codex-mini": { - "id": "gpt-5.1-codex-mini", - "name": "GPT-5.1 Codex mini", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.25, - "output": 2, - "cacheRead": 0.025, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 19, - "thinking": { - "mode": "effort", - "efforts": [ - "medium", - "high" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.2": { - "id": "gpt-5.2", - "name": "GPT-5.2", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.75, - "output": 14, - "cacheRead": 0.175, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 29, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.2-codex": { - "id": "gpt-5.2-codex", - "name": "GPT-5.2 Codex", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.75, - "output": 14, - "cacheRead": 0.175, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 8, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - }, - "applyPatchToolType": "freeform" - }, - "gpt-5.3-codex": { - "id": "gpt-5.3-codex", - "name": "GPT-5.3 Codex", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.75, - "output": 14, - "cacheRead": 0.175, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 25, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - }, - "applyPatchToolType": "freeform" - }, "gpt-5.3-codex-spark": { "id": "gpt-5.3-codex-spark", "name": "GPT-5.3 Codex Spark", @@ -63354,7 +62710,6 @@ "provider": "openai-codex", "baseUrl": "https://chatgpt.com/backend-api", "reasoning": true, - "preferWebsockets": true, "input": [ "text" ], @@ -63364,9 +62719,16 @@ "cacheRead": 0.175, "cacheWrite": 0 }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, "contextWindow": 128000, "maxTokens": 128000, - "contextPromotionTarget": "openai-codex/gpt-5.5", + "preferWebsockets": true, + "priority": 26, + "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ @@ -63376,7 +62738,7 @@ "xhigh" ] }, - "applyPatchToolType": "freeform" + "contextPromotionTarget": "openai-codex/gpt-5.5" }, "gpt-5.4": { "id": "gpt-5.4", @@ -63452,38 +62814,6 @@ ] } }, - "gpt-5.4-nano": { - "id": "gpt-5.4-nano", - "name": "GPT-5.4 nano", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.2, - "output": 1.25, - "cacheRead": 0.02, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 2, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - }, - "applyPatchToolType": "freeform" - }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", @@ -63831,9 +63161,9 @@ "text" ], "cost": { - "input": 1.74, - "output": 3.48, - "cacheRead": 0.0145, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.003625, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -63939,6 +63269,65 @@ ] } }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-responses", + "provider": "opencode-go", + "baseUrl": "https://opencode.ai/zen/go/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "hy3": { + "id": "hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "opencode-go", + "baseUrl": "https://opencode.ai/zen/go/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.58, + "cacheRead": 0.035, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", @@ -64032,6 +63421,39 @@ ] } }, + "kimi-k3": { + "id": "kimi-k3", + "name": "Kimi K3 (2x usage)", + "api": "openai-completions", + "provider": "opencode-go", + "baseUrl": "https://opencode.ai/zen/go/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "mimo-v2-omni": { "id": "mimo-v2-omni", "name": "MiMo-V2-Omni", @@ -64135,9 +63557,9 @@ "text" ], "cost": { - "input": 1.74, - "output": 3.48, - "cacheRead": 0.0145, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.003625, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -64935,9 +64357,69 @@ "requiresEffort": true } }, + "gemini-3.5-flash-lite": { + "id": "gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "google-generative-ai", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "gemini-3.6-flash": { + "id": "gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "google-generative-ai", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "glm-4.6": { "id": "glm-4.6", - "name": "glm-4.6", + "name": "GLM-4.6", "api": "openai-completions", "provider": "opencode-zen", "baseUrl": "https://opencode.ai/zen/v1", @@ -65666,7 +65148,7 @@ "grok-4.5": { "id": "grok-4.5", "name": "Grok 4.5", - "api": "openai-completions", + "api": "openai-responses", "provider": "opencode-zen", "baseUrl": "https://opencode.ai/zen/v1", "reasoning": true, @@ -65908,6 +65390,35 @@ ] } }, + "laguna-s-2.1-free": { + "id": "laguna-s-2.1-free", + "name": "Laguna S 2.1 Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 32000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "ling-2.6-flash-free": { "id": "ling-2.6-flash-free", "name": "Ling 2.6 Flash Free", @@ -66568,7 +66079,7 @@ ], "cost": { "input": 1.5, - "output": 9, + "output": 7.5, "cacheRead": 0.15, "cacheWrite": 0.08333333333333334 }, @@ -66625,12 +66136,12 @@ "image" ], "cost": { - "input": 0.66, - "output": 3.41, - "cacheRead": 0.15, + "input": 3, + "output": 15, + "cacheRead": 0.3, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 1048576, "maxTokens": 262144, "thinking": { "mode": "effort", @@ -66714,7 +66225,7 @@ "cost": { "input": 2, "output": 6, - "cacheRead": 0.5, + "cacheRead": 0.3, "cacheWrite": 0 }, "contextWindow": 500000, @@ -68041,7 +67552,7 @@ "cacheRead": 0.15, "cacheWrite": 0 }, - "contextWindow": 131072, + "contextWindow": 163840, "maxTokens": 16000 }, "deepseek/deepseek-chat-v3-0324": { @@ -68055,13 +67566,13 @@ "text" ], "cost": { - "input": 0.24, - "output": 0.8999999999999999, + "input": 0.27, + "output": 1.12, "cacheRead": 0.135, "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 16384 + "maxTokens": 65536 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -68074,8 +67585,8 @@ "text" ], "cost": { - "input": 0.21, - "output": 0.7899999999999999, + "input": 0.25, + "output": 0.95, "cacheRead": 0.13, "cacheWrite": 0 }, @@ -68150,8 +67661,8 @@ ], "cost": { "input": 0.27, - "output": 0.95, - "cacheRead": 0.13, + "output": 1, + "cacheRead": 0.135, "cacheWrite": 0 }, "contextWindow": 163840, @@ -68199,13 +67710,13 @@ "text" ], "cost": { - "input": 0.2145, - "output": 0.32175, - "cacheRead": 0.02145, + "input": 0.26899999999999996, + "output": 0.39999999999999997, + "cacheRead": 0.13449999999999998, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 64000, + "contextWindow": 163840, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -68249,9 +67760,9 @@ "text" ], "cost": { - "input": 0.08399999999999999, - "output": 0.16799999999999998, - "cacheRead": 0.016800000000000002, + "input": 0.098, + "output": 0.196, + "cacheRead": 0.0196, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -68625,7 +68136,7 @@ "cacheRead": 0.19999999999999998, "cacheWrite": 0.375 }, - "contextWindow": 65536, + "contextWindow": 131072, "maxTokens": 32768, "thinking": { "mode": "effort", @@ -68759,7 +68270,7 @@ "cacheRead": 0.19999999999999998, "cacheWrite": 0.375 }, - "contextWindow": 1048756, + "contextWindow": 1048576, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -68800,6 +68311,66 @@ "requiresEffort": true } }, + "google/gemini-3.5-flash-lite": { + "id": "google/gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.03, + "cacheWrite": 0.08333333333333334 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "google/gemini-3.6-flash": { + "id": "google/gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0.08333333333333334 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", "name": "Gemma 3 12B", @@ -68832,13 +68403,13 @@ "image" ], "cost": { - "input": 0.08, - "output": 0.16, - "cacheRead": 0.015, + "input": 0.09999999999999999, + "output": 0.3, + "cacheRead": 0.04, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 16384 + "contextWindow": 262144, + "maxTokens": 131072 }, "google/gemma-3-27b-it:free": { "id": "google/gemma-3-27b-it:free", @@ -68872,13 +68443,13 @@ "image" ], "cost": { - "input": 0.06, - "output": 0.33, + "input": 0.07, + "output": 0.33999999999999997, "cacheRead": 0.04, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -68965,7 +68536,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 8192, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -69193,6 +68764,25 @@ ] } }, + "kwaipilot/kat-coder-air-v2.5": { + "id": "kwaipilot/kat-coder-air-v2.5", + "name": "KAT-Coder-Air V2.5", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 0.6, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 80000 + }, "kwaipilot/kat-coder-pro": { "id": "kwaipilot/kat-coder-pro", "name": "KAT-Coder-Pro V1", @@ -69228,6 +68818,25 @@ "cacheRead": 0.06, "cacheWrite": 0 }, + "contextWindow": 262144, + "maxTokens": 80000 + }, + "kwaipilot/kat-coder-pro-v2.5": { + "id": "kwaipilot/kat-coder-pro-v2.5", + "name": "KAT-Coder-Pro V2.5", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.74, + "output": 2.96, + "cacheRead": 0.15, + "cacheWrite": 0 + }, "contextWindow": 256000, "maxTokens": 80000 }, @@ -69260,6 +68869,34 @@ "requiresEffort": true } }, + "meituan/longcat-2.0": { + "id": "meituan/longcat-2.0", + "name": "LongCat 2.0", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.006, + "cacheWrite": 0 + }, + "contextWindow": 1048756, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "meituan/longcat-flash-chat": { "id": "meituan/longcat-flash-chat", "name": "LongCat Flash Chat", @@ -69347,13 +68984,13 @@ "text" ], "cost": { - "input": 0.02, - "output": 0.03, - "cacheRead": 0, + "input": 0.049999999999999996, + "output": 0.08, + "cacheRead": 0.024999999999999998, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 16384 + "maxTokens": 131072 }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", @@ -69366,13 +69003,13 @@ "text" ], "cost": { - "input": 0.09999999999999999, - "output": 0.32, + "input": 0.13, + "output": 0.39999999999999997, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 16384 + "maxTokens": 128000 }, "meta-llama/llama-3.3-70b-instruct:free": { "id": "meta-llama/llama-3.3-70b-instruct:free", @@ -69405,8 +69042,8 @@ "image" ], "cost": { - "input": 0.15, - "output": 0.6, + "input": 0.19999999999999998, + "output": 0.7999999999999999, "cacheRead": 0, "cacheWrite": 0 }, @@ -69430,9 +69067,38 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 10000000, + "contextWindow": 1310720, "maxTokens": 16384 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "Muse Spark 1.1", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 4.25, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "minimax/minimax-m1": { "id": "minimax/minimax-m1", "name": "MiniMax M1", @@ -69444,7 +69110,7 @@ "text" ], "cost": { - "input": 0.39999999999999997, + "input": 0.55, "output": 2.2, "cacheRead": 0, "cacheWrite": 0 @@ -69587,13 +69253,13 @@ "text" ], "cost": { - "input": 0.24, - "output": 0.96, + "input": 0.25, + "output": 1, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 204800, - "maxTokens": 196608, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -69622,7 +69288,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 512000, "thinking": { "mode": "effort", "efforts": [ @@ -69926,13 +69592,13 @@ "text" ], "cost": { - "input": 0.02, + "input": 0.019000000000000003, "output": 0.03, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 128000 + "maxTokens": 16384 }, "mistralai/mistral-saba": { "id": "mistralai/mistral-saba", @@ -70053,12 +69719,12 @@ "image" ], "cost": { - "input": 0.075, - "output": 0.19999999999999998, - "cacheRead": 0.03, + "input": 0.09999999999999999, + "output": 0.3, + "cacheRead": 0.01, "cacheWrite": 0 }, - "contextWindow": 128000, + "contextWindow": 256000, "maxTokens": 16384 }, "mistralai/mistral-small-creative": { @@ -70255,13 +69921,13 @@ "image" ], "cost": { - "input": 0.375, - "output": 2.025, - "cacheRead": 0.203, + "input": 0.5700000000000001, + "output": 2.8499999999999996, + "cacheRead": 0.095, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 64000, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -70284,9 +69950,9 @@ "image" ], "cost": { - "input": 0.66, - "output": 3.41, - "cacheRead": 0.15, + "input": 0.684, + "output": 3.42, + "cacheRead": 0.144, "cacheWrite": 0 }, "contextWindow": 262144, @@ -70342,9 +70008,9 @@ "image" ], "cost": { - "input": 0.72, - "output": 3.49, - "cacheRead": 0.159, + "input": 0.82, + "output": 3.75, + "cacheRead": 0.16, "cacheWrite": 0 }, "contextWindow": 262144, @@ -70359,6 +70025,39 @@ ] } }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "nex-agi/deepseek-v3.1-nex-n1": { "id": "nex-agi/deepseek-v3.1-nex-n1", "name": "DeepSeek V3.1 Nex N1", @@ -70667,7 +70366,7 @@ "cost": { "input": 0.08, "output": 0.44999999999999996, - "cacheRead": 0.09999999999999999, + "cacheRead": 0.06, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -70698,7 +70397,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 1000000, + "contextWindow": 262144, "maxTokens": 262144, "thinking": { "mode": "effort", @@ -70721,12 +70420,12 @@ "text" ], "cost": { - "input": 0.5, - "output": 2.2, - "cacheRead": 0.09999999999999999, + "input": 0.6, + "output": 3.5999999999999996, + "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, - "contextWindow": 1000000, + "contextWindow": 512288, "maxTokens": 16384, "thinking": { "mode": "effort", @@ -71382,7 +71081,7 @@ "cost": { "input": 0.049999999999999996, "output": 0.39999999999999997, - "cacheRead": 0.01, + "cacheRead": 0.005, "cacheWrite": 0 }, "contextWindow": 400000, @@ -71440,7 +71139,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.13, + "cacheRead": 0.125, "cacheWrite": 0 }, "contextWindow": 400000, @@ -71469,11 +71168,11 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.13, + "cacheRead": 0.125, "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 32000 + "maxTokens": 16384 }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", @@ -71489,7 +71188,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.13, + "cacheRead": 0.125, "cacheWrite": 0 }, "contextWindow": 272000, @@ -72150,8 +71849,8 @@ "text" ], "cost": { - "input": 0.036, - "output": 0.18, + "input": 0.037, + "output": 0.16999999999999998, "cacheRead": 0, "cacheWrite": 0 }, @@ -72231,13 +71930,13 @@ "text" ], "cost": { - "input": 0.029, - "output": 0.14, - "cacheRead": 0.015, + "input": 0.03, + "output": 0.13, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 65536, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -72616,6 +72315,35 @@ ] } }, + "openrouter/auto-beta": { + "id": "openrouter/auto-beta", + "name": "Auto Router (Beta)", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": -1000000, + "output": -1000000, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 2000000, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "openrouter/elephant-alpha": { "id": "openrouter/elephant-alpha", "name": "Elephant", @@ -72808,6 +72536,62 @@ ] } }, + "poolside/laguna-s-2.1": { + "id": "poolside/laguna-s-2.1", + "name": "Laguna S 2.1", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.09999999999999999, + "output": 0.19999999999999998, + "cacheRead": 0.01, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "poolside/laguna-s-2.1:free": { + "id": "poolside/laguna-s-2.1:free", + "name": "Laguna S 2.1 (free)", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "poolside/laguna-xs-2.1": { "id": "poolside/laguna-xs-2.1", "name": "Laguna XS 2.1", @@ -72964,7 +72748,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, + "contextWindow": 32768, "maxTokens": 16384 }, "qwen/qwen-2.5-7b-instruct": { @@ -72983,7 +72767,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, + "contextWindow": 32768, "maxTokens": 32768 }, "qwen/qwen-max": { @@ -73121,13 +72905,13 @@ "text" ], "cost": { - "input": 0.09999999999999999, - "output": 0.24, + "input": 0.22749999999999998, + "output": 0.9099999999999999, "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131702, - "maxTokens": 40960, + "contextWindow": 131072, + "maxTokens": 8192, "thinking": { "mode": "effort", "efforts": [ @@ -73178,7 +72962,7 @@ ], "cost": { "input": 0.09, - "output": 0.09999999999999999, + "output": 0.55, "cacheRead": 0, "cacheWrite": 0 }, @@ -73196,13 +72980,13 @@ "text" ], "cost": { - "input": 0.14950000000000002, - "output": 1.495, + "input": 0.3, + "output": 3, "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -73225,13 +73009,13 @@ "text" ], "cost": { - "input": 0.12, - "output": 0.5, + "input": 0.13, + "output": 0.52, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 16384, + "maxTokens": 8192, "thinking": { "mode": "effort", "efforts": [ @@ -73253,12 +73037,12 @@ "text" ], "cost": { - "input": 0.04815, - "output": 0.19305, + "input": 0.09999999999999999, + "output": 0.3, "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, + "contextWindow": 262144, "maxTokens": 32000 }, "qwen/qwen3-30b-a3b-thinking-2507": { @@ -73277,7 +73061,7 @@ "cacheRead": 0.08, "cacheWrite": 0 }, - "contextWindow": 131072, + "contextWindow": 81920, "maxTokens": 32768, "thinking": { "mode": "effort", @@ -73413,12 +73197,12 @@ "text" ], "cost": { - "input": 0.22, - "output": 1.7999999999999998, - "cacheRead": 0.022, + "input": 0.3, + "output": 1, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 262144, "maxTokens": 65536 }, "qwen/qwen3-coder-30b-a3b-instruct": { @@ -73437,7 +73221,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 160000, + "contextWindow": 262144, "maxTokens": 32768 }, "qwen/qwen3-coder-flash": { @@ -73594,13 +73378,13 @@ "text" ], "cost": { - "input": 0.09, + "input": 0.09999999999999999, "output": 1.1, - "cacheRead": 0, + "cacheRead": 0.07, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384 + "maxTokens": 262144 }, "qwen/qwen3-next-80b-a3b-instruct:free": { "id": "qwen/qwen3-next-80b-a3b-instruct:free", @@ -73662,13 +73446,13 @@ "image" ], "cost": { - "input": 0.19999999999999998, - "output": 0.88, - "cacheRead": 0.11, + "input": 0.21, + "output": 1.9, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384 + "maxTokens": 32768 }, "qwen/qwen3-vl-235b-a22b-thinking": { "id": "qwen/qwen3-vl-235b-a22b-thinking", @@ -73737,7 +73521,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, + "contextWindow": 262144, "maxTokens": 32768, "thinking": { "mode": "effort", @@ -73767,7 +73551,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 131072, "maxTokens": 32768 }, "qwen/qwen3-vl-8b-instruct": { @@ -73787,7 +73571,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 256000, + "contextWindow": 262144, "maxTokens": 32768 }, "qwen/qwen3-vl-8b-thinking": { @@ -73807,7 +73591,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 256000, + "contextWindow": 131072, "maxTokens": 32768, "thinking": { "mode": "effort", @@ -73838,7 +73622,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -73861,13 +73645,13 @@ "image" ], "cost": { - "input": 0.195, - "output": 1.56, + "input": 0.26, + "output": 2.6, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 81920, "thinking": { "mode": "effort", "efforts": [ @@ -73896,7 +73680,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 81920, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -73919,12 +73703,12 @@ "image" ], "cost": { - "input": 0.385, - "output": 2.4499999999999997, - "cacheRead": 0.111, + "input": 0.39, + "output": 2.34, + "cacheRead": 0.22499999999999998, "cacheWrite": 0 }, - "contextWindow": 256000, + "contextWindow": 262144, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -74064,13 +73848,13 @@ "image" ], "cost": { - "input": 0.28500000000000003, - "output": 2.4, + "input": 0.44999999999999996, + "output": 2.7, "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262140, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -74264,10 +74048,10 @@ "text" ], "cost": { - "input": 1.25, - "output": 3.75, - "cacheRead": 0.25, - "cacheWrite": 1.5625 + "input": 1.475, + "output": 4.425, + "cacheRead": 0.295, + "cacheWrite": 1.84375 }, "contextWindow": 1000000, "maxTokens": 65536, @@ -74543,7 +74327,7 @@ "cacheRead": 0.04, "cacheWrite": 0 }, - "contextWindow": 256000, + "contextWindow": 262144, "maxTokens": 256000, "thinking": { "mode": "effort", @@ -74708,6 +74492,35 @@ "contextWindow": 32768, "maxTokens": 32768 }, + "thinkingmachines/inkling": { + "id": "thinkingmachines/inkling", + "name": "Inkling", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 4.05, + "cacheRead": 0.16999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 524288, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "tngtech/deepseek-r1t2-chimera": { "id": "tngtech/deepseek-r1t2-chimera", "name": "DeepSeek R1T2 Chimera", @@ -75099,7 +74912,7 @@ "cost": { "input": 2, "output": 6, - "cacheRead": 0.5, + "cacheRead": 0.3, "cacheWrite": 0 }, "contextWindow": 500000, @@ -75265,12 +75078,12 @@ "image" ], "cost": { - "input": 0.105, + "input": 0.14, "output": 0.28, - "cacheRead": 0.028, + "cacheRead": 0.0028, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 1050000, "maxTokens": 131072, "thinking": { "mode": "effort", @@ -75297,7 +75110,7 @@ "cacheRead": 0.0036, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 1050000, "maxTokens": 131072, "thinking": { "mode": "effort", @@ -75451,12 +75264,12 @@ "text" ], "cost": { - "input": 0.43, - "output": 1.74, - "cacheRead": 0.08, + "input": 0.5, + "output": 2, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, - "contextWindow": 202752, + "contextWindow": 204800, "maxTokens": 131072, "thinking": { "mode": "effort", @@ -75541,7 +75354,7 @@ "cacheRead": 0.08, "cacheWrite": 0 }, - "contextWindow": 202752, + "contextWindow": 204800, "maxTokens": 131072, "thinking": { "mode": "effort", @@ -75564,13 +75377,13 @@ "text" ], "cost": { - "input": 0.06, + "input": 0.060500000000000005, "output": 0.39999999999999997, "cacheRead": 0.01, "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 16384, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -75592,13 +75405,13 @@ "text" ], "cost": { - "input": 0.6, - "output": 1.92, - "cacheRead": 0.12, + "input": 0.95, + "output": 2.5500000000000003, + "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, - "contextWindow": 202752, - "maxTokens": 128000, + "contextWindow": 204800, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -75625,7 +75438,7 @@ "cacheRead": 0.24, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 202752, "maxTokens": 131072 }, "z-ai/glm-5.1": { @@ -75644,7 +75457,7 @@ "cacheRead": 0.1794, "cacheWrite": 0 }, - "contextWindow": 202752, + "contextWindow": 204800, "maxTokens": 128000, "thinking": { "mode": "effort", @@ -75667,9 +75480,9 @@ "text" ], "cost": { - "input": 0.42, - "output": 1.32, - "cacheRead": 0.078, + "input": 0.7756000000000001, + "output": 2.4376, + "cacheRead": 0.14404, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -75709,7 +75522,7 @@ "qianfan": { "deepseek-v3.2": { "id": "deepseek-v3.2", - "name": "deepseek-v3.2", + "name": "DeepSeek V3.2", "api": "openai-completions", "provider": "qianfan", "baseUrl": "https://qianfan.baidubce.com/v2", @@ -76831,6 +76644,36 @@ "contextWindow": 1000000, "maxTokens": 500000 }, + "thinkingmachines/Inkling": { + "id": "thinkingmachines/Inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "together", + "baseUrl": "https://api.together.xyz/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 4.05, + "cacheRead": 0.17, + "cacheWrite": 0 + }, + "contextWindow": 524288, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM-4.7", @@ -76961,9 +76804,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.95, + "output": 4, + "cacheRead": 0.19, "cacheWrite": 0 }, "contextWindow": 262144, @@ -76995,9 +76838,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.15, + "output": 1, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 262144, @@ -77034,9 +76877,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, "contextWindow": 405504, @@ -77057,9 +76900,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.95, + "output": 4, + "cacheRead": 0.19, "cacheWrite": 0 }, "contextWindow": 262144, @@ -77091,9 +76934,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.15, + "output": 1, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 262144, @@ -78067,7 +77910,7 @@ "cacheWrite": 0 }, "contextWindow": 32000, - "maxTokens": null, + "maxTokens": 65536, "compat": { "supportsUsageInStreaming": false } @@ -78198,6 +78041,66 @@ ] } }, + "gemini-3-5-flash-lite": { + "id": "gemini-3-5-flash-lite", + "name": "Gemini 3.5 Flash-Lite", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.375, + "output": 3.125, + "cacheRead": 0.0375, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gemini-3-6-flash": { + "id": "gemini-3-6-flash", + "name": "Gemini 3.6 Flash", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.875, + "output": 9.375, + "cacheRead": 0.1875, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", @@ -78543,7 +78446,7 @@ "cost": { "input": 2.27, "output": 6.8, - "cacheRead": 0.57, + "cacheRead": 0.34, "cacheWrite": 0 }, "contextWindow": 500000, @@ -78676,6 +78579,36 @@ "supportsUsageInStreaming": false } }, + "inkling": { + "id": "inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 5.0625, + "cacheRead": 0.2125, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "kimi-k2-5": { "id": "kimi-k2-5", "name": "Kimi K2.5", @@ -78799,6 +78732,39 @@ "supportsUsageInStreaming": false } }, + "kimi-k3": { + "id": "kimi-k3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3.75, + "output": 18.75, + "cacheRead": 0.375, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "llama-3.2-3b": { "id": "llama-3.2-3b", "name": "Llama 3.2 3B", @@ -79961,6 +79927,35 @@ ] } }, + "qwen3-6-35b-a3b": { + "id": "qwen3-6-35b-a3b", + "name": "Qwen 3.6 35B A3B", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.15, + "output": 1, + "cacheRead": 0.05, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen 3 Coder 480b", @@ -80352,7 +80347,7 @@ "cacheWrite": 0 }, "contextWindow": 200000, - "maxTokens": 24000, + "maxTokens": 80000, "thinking": { "mode": "effort", "efforts": [ @@ -81315,7 +81310,7 @@ "cacheWrite": 18.75 }, "contextWindow": 200000, - "maxTokens": 32000, + "maxTokens": 8192, "thinking": { "mode": "budget", "efforts": [ @@ -81447,6 +81442,37 @@ "supportsDisplay": true } }, + "anthropic/claude-opus-4.7-fast": { + "id": "anthropic/claude-opus-4.7-fast", + "name": "Claude Opus 4.7 (Fast)", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 30, + "output": 150, + "cacheRead": 3, + "cacheWrite": 37.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "supportsDisplay": true + } + }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", @@ -81478,6 +81504,37 @@ "supportsDisplay": true } }, + "anthropic/claude-opus-4.8-fast": { + "id": "anthropic/claude-opus-4.8-fast", + "name": "Claude Opus 4.8 (Fast)", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 50, + "cacheRead": 1, + "cacheWrite": 12.5 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ], + "supportsDisplay": true + } + }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", @@ -81496,7 +81553,7 @@ "cacheWrite": 3.75 }, "contextWindow": 1000000, - "maxTokens": 64000, + "maxTokens": 8192, "thinking": { "mode": "budget", "efforts": [ @@ -81813,12 +81870,12 @@ "text" ], "cost": { - "input": 0.6, - "output": 1.7, - "cacheRead": 0.28, + "input": 0.25, + "output": 0.95, + "cacheRead": 0.13, "cacheWrite": 0 }, - "contextWindow": 128000, + "contextWindow": 163840, "maxTokens": 128000, "thinking": { "mode": "budget", @@ -82330,6 +82387,66 @@ "requiresEffort": true } }, + "google/gemini-3.5-flash-lite": { + "id": "google/gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "google/gemini-3.6-flash": { + "id": "google/gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B", @@ -82468,6 +82585,35 @@ ] } }, + "kwaipilot/kat-coder-air-v2.5": { + "id": "kwaipilot/kat-coder-air-v2.5", + "name": "Kat Coder Air V2.5", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 0.6, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 80000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "kwaipilot/kat-coder-pro-v1": { "id": "kwaipilot/kat-coder-pro-v1", "name": "KAT-Coder-Pro V1", @@ -82506,6 +82652,35 @@ "contextWindow": 256000, "maxTokens": 256000 }, + "kwaipilot/kat-coder-pro-v2.5": { + "id": "kwaipilot/kat-coder-pro-v2.5", + "name": "Kat Coder Pro V2.5", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.74, + "output": 2.96, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 80000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "meituan/longcat-flash-chat": { "id": "meituan/longcat-flash-chat", "name": "LongCat Flash Chat", @@ -83589,6 +83764,36 @@ ] } }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "requiresEffort": true + } + }, "nvidia/nemotron-3-nano-30b-a3b": { "id": "nvidia/nemotron-3-nano-30b-a3b", "name": "Nemotron 3 Nano 30B A3B", @@ -84556,7 +84761,7 @@ }, "openai/gpt-5.6-luna": { "id": "openai/gpt-5.6-luna", - "name": "GPT 5.6 Luna", + "name": "GPT-5.6 Luna", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -84586,7 +84791,7 @@ }, "openai/gpt-5.6-sol": { "id": "openai/gpt-5.6-sol", - "name": "GPT 5.6 Sol", + "name": "GPT-5.6 Sol", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -84616,7 +84821,7 @@ }, "openai/gpt-5.6-terra": { "id": "openai/gpt-5.6-terra", - "name": "GPT 5.6 Terra", + "name": "GPT-5.6 Terra", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -84956,6 +85161,64 @@ "contextWindow": 200000, "maxTokens": 8000 }, + "poolside/laguna-s-2.1": { + "id": "poolside/laguna-s-2.1", + "name": "Laguna S 2.1", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.09999999999999999, + "output": 0.19999999999999998, + "cacheRead": 0.01, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "poolside/laguna-s-2.1-free": { + "id": "poolside/laguna-s-2.1-free", + "name": "Laguna S 2.1 Free", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 32768, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "prime-intellect/intellect-3": { "id": "prime-intellect/intellect-3", "name": "INTELLECT-3", @@ -85074,6 +85337,65 @@ ] } }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.58, + "cacheRead": 0.035, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "thinkingmachines/inkling": { + "id": "thinkingmachines/inkling", + "name": "Inkling", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 4.05, + "cacheRead": 0.16999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 256000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "vercel/v0-1.0-md": { "id": "vercel/v0-1.0-md", "name": "v0-1.0-md", @@ -85548,7 +85870,7 @@ "cost": { "input": 2, "output": 6, - "cacheRead": 0.5, + "cacheRead": 0.3, "cacheWrite": 0 }, "contextWindow": 500000, @@ -86132,9 +86454,9 @@ "text" ], "cost": { - "input": 3, - "output": 10.25, - "cacheRead": 0.5, + "input": 2.0999999999999996, + "output": 6.6000000000000005, + "cacheRead": 0.21, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -86278,9 +86600,9 @@ "text" ], "cost": { - "input": 1.5, - "output": 5.125, - "cacheRead": 0.25, + "input": 1.75, + "output": 5.5, + "cacheRead": 0.325, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -87161,7 +87483,7 @@ "cost": { "input": 2, "output": 6, - "cacheRead": 0.5, + "cacheRead": 0.3, "cacheWrite": 0 }, "contextWindow": 500000, @@ -87299,9 +87621,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": true, - "supportsReasoningEffort": false, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": false } }, "grok-4.20-0309-reasoning": { @@ -87329,9 +87651,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": true, - "supportsReasoningEffort": false, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": false } }, "grok-4.20-multi-agent-0309": { @@ -87371,9 +87693,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": false, - "supportsReasoningEffort": true, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": true } }, "grok-4.3": { @@ -87414,9 +87736,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": false, - "supportsReasoningEffort": true, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": true } }, "grok-4.5": { @@ -87457,9 +87779,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": false, - "supportsReasoningEffort": true, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": true } }, "grok-build": { @@ -87487,9 +87809,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": true, - "supportsReasoningEffort": false, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": false } }, "grok-build-0.1": { @@ -87517,9 +87839,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": true, - "supportsReasoningEffort": false, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": false } }, "grok-composer-2.5-fast": { @@ -87546,9 +87868,9 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, "omitReasoningEffort": true, - "supportsReasoningEffort": false, - "supportsImageDetailOriginal": false + "supportsReasoningEffort": false } } }, @@ -88126,9 +88448,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 131072, @@ -88155,9 +88477,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.2, + "output": 1.1, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 131072, @@ -88214,8 +88536,8 @@ "image" ], "cost": { - "input": 0, - "output": 0, + "input": 0.6, + "output": 1.8, "cacheRead": 0, "cacheWrite": 0 }, @@ -88234,7 +88556,7 @@ }, "glm-4.6": { "id": "glm-4.6", - "name": "glm-4.6", + "name": "GLM-4.6", "api": "anthropic-messages", "provider": "zai", "baseUrl": "https://api.z.ai/api/anthropic", @@ -88243,9 +88565,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 204800, @@ -88273,8 +88595,8 @@ "image" ], "cost": { - "input": 0, - "output": 0, + "input": 0.3, + "output": 0.9, "cacheRead": 0, "cacheWrite": 0 }, @@ -88302,9 +88624,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 204800, @@ -88389,9 +88711,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1, + "output": 3.2, + "cacheRead": 0.2, "cacheWrite": 0 }, "contextWindow": 204800, @@ -88418,9 +88740,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.2, + "output": 4, + "cacheRead": 0.24, "cacheWrite": 0 }, "contextWindow": 200000, @@ -88447,9 +88769,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, "contextWindow": 200000, @@ -88476,9 +88798,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -88503,9 +88825,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.2, + "output": 4, + "cacheRead": 0.24, "cacheWrite": 0 }, "contextWindow": 200000, @@ -89992,6 +90314,66 @@ "requiresEffort": true } }, + "google/gemini-3.5-flash-lite": { + "id": "google/gemini-3.5-flash-lite", + "name": "Gemini 3.5 Flash Lite", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "google/gemini-3.6-flash": { + "id": "google/gemini-3.6-flash", + "name": "Gemini 3.6 Flash", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "google/gemini-embedding-2": { "id": "google/gemini-embedding-2", "name": "Gemini Embedding 2", @@ -90948,6 +91330,72 @@ ] } }, + "moonshotai/kimi-k3": { + "id": "moonshotai/kimi-k3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, + "moonshotai/kimi-k3-free": { + "id": "moonshotai/kimi-k3-free", + "name": "Kimi K3 (Free)", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "openai/chat-latest": { "id": "openai/chat-latest", "name": "Chat Latest (GPT-5.5 Instant)", @@ -93784,7 +94232,7 @@ "zhipu-coding-plan": { "glm-4.5": { "id": "glm-4.5", - "name": "glm-4.5", + "name": "GLM-4.5", "api": "openai-completions", "provider": "zhipu-coding-plan", "baseUrl": "https://open.bigmodel.cn/api/coding/paas/v4", @@ -93845,7 +94293,7 @@ }, "glm-4.6": { "id": "glm-4.6", - "name": "glm-4.6", + "name": "GLM-4.6", "api": "openai-completions", "provider": "zhipu-coding-plan", "baseUrl": "https://open.bigmodel.cn/api/coding/paas/v4", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index f358bf4e5..d4fe73a53 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1,13 +1,15 @@ +import * as logger from "@oh-my-pi/pi-utils/logger"; import { fetchOpenAICompatibleModels, type OpenAICompatibleModelMapperContext, type OpenAICompatibleModelRecord, } from "../discovery/openai-compatible"; -import { Effort } from "../effort"; +import { Effort, THINKING_EFFORTS } from "../effort"; import { FIREWORKS_FAST_SUFFIX, toFireworksPublicModelId } from "../fireworks-model-id"; import { isGlmVisionModelId, isGrokReasoningEffortCapable, + isKimiK3ModelId, isKimiModelId, isReasoningGlmModelId, } from "../identity/family"; @@ -1494,13 +1496,31 @@ export function zhipuCodingPlanModelManagerOptions( export const FIREWORKS_KIMI_MAX_TOKENS = 32_768; /** - * Returns true for any Kimi K2.x public model id served by Fireworks-backed - * providers (`fireworks` direct, `firepass` router). Matches both the public - * catalog id (`kimi-k2.5`, `kimi-k2.6`, `kimi-k2.6-turbo`) and the canonical - * Fireworks wire id (`accounts/fireworks/{models,routers}/kimi-k2…`). + * Fireworks' output ceiling for Kimi K2.7-Code specifically. Its `/v1/models` + * generic `max_completion_tokens` is 65,536 and Fireworks serves it in full — + * verified with a single completion emitting 58,971 output tokens and + * `max_tokens: 200000` accepted without error. Unlike the older K2.5/K2.6 + * family (see {@link FIREWORKS_KIMI_MAX_TOKENS}), K2.7-Code is not clamped to + * 32,768; that ceiling only truncated it. + */ +export const FIREWORKS_KIMI_K27_CODE_MAX_TOKENS = 65_536; + +/** + * Returns true for the Kimi K2.5 / K2.6 family served by Fireworks-backed + * providers (`fireworks` direct, `firepass` router) that share the 32,768 + * `maxTokens` ceiling. Matches both the public catalog id (`kimi-k2.5`, + * `kimi-k2.6`, `kimi-k2.6-turbo`) and the canonical Fireworks wire id + * (`accounts/fireworks/{models,routers}/kimi-k2…`). + * + * K2.7-Code (incl. `-fast` / `-highspeed`) is deliberately excluded: unlike the + * earlier K2 family it serves its full context on Fireworks — verified with a + * single completion emitting 58,971 output tokens and `max_tokens: 200000` + * accepted without error — so the 32,768 cap would only truncate it. It inherits + * Fireworks' reported `max_completion_tokens` (65,536) instead. */ export function isFireworksKimiK2ModelId(modelId: string): boolean { const trimmed = modelId.toLowerCase(); + if (/kimi[-._]?k2(?:[._-]?|p)7[-._]?code/.test(trimmed)) return false; if (trimmed.startsWith("kimi-k2")) return true; return /\/kimi-k2(?:p\d+)?(?:[._-]|$)/.test(trimmed); } @@ -1657,9 +1677,14 @@ function mapFireworksControlPlaneModel( const supportsImage = toBoolean(record.supportsImageInput) === true; const supportsTools = toBoolean(record.supportsTools); const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? null); - // The control plane reports no max-output budget; default the Kimi family to - // its published cap, everyone else to the discovery fallback, then clamp. - const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : null; + // The control plane reports no max-output budget. Default K2.7-Code to its + // verified 65,536 ceiling, the older K2.5/K2.6 family to the clamped 32,768, + // everyone else to the discovery fallback, then clamp. + const fallbackMaxTokens = isKimiK27CodeModelId(publicModelId) + ? FIREWORKS_KIMI_K27_CODE_MAX_TOKENS + : isFireworksKimiK2ModelId(publicModelId) + ? FIREWORKS_KIMI_MAX_TOKENS + : null; const maxTokens = clampFireworksKimiMaxTokens(publicModelId, reference?.maxTokens ?? fallbackMaxTokens); const base: ModelSpec<"openai-completions"> = reference ?? { id: publicModelId, @@ -2480,6 +2505,45 @@ export interface KimiCodeModelManagerConfig { fetch?: FetchImpl; } +function mapKimiThinking(entry: OpenAICompatibleModelRecord): ThinkingConfig | undefined { + const raw = entry.think_efforts; + if (!isRecord(raw) || raw.support !== true) return undefined; + const validEfforts = raw.valid_efforts; + if (!Array.isArray(validEfforts)) return undefined; + const efforts = THINKING_EFFORTS.filter(effort => validEfforts.includes(effort)); + if (efforts.length === 0) return undefined; + + const thinking: ThinkingConfig = { mode: "effort", efforts }; + if (entry.supports_thinking_type === "only") { + thinking.requiresEffort = true; + } + if (typeof raw.default_effort === "string") { + const defaultLevel = THINKING_EFFORTS.find(effort => effort === raw.default_effort); + if (defaultLevel !== undefined && efforts.includes(defaultLevel)) { + thinking.defaultLevel = defaultLevel; + } + } + return thinking; +} + +function kimiSupportsReasoning(entry: OpenAICompatibleModelRecord, modelId: string): boolean { + switch (entry.supports_thinking_type) { + case "only": + case "both": + return true; + case "no": + return false; + default: + return entry.supports_reasoning === true || modelId.includes("thinking"); + } +} + +function mapKimiApiFormat(protocol: unknown): OpenAICompat["kimiApiFormat"] { + if (protocol === "anthropic") return "anthropic"; + if (protocol === null) return "openai"; + return undefined; +} + export function kimiCodeModelManagerOptions( config?: KimiCodeModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { @@ -2504,15 +2568,19 @@ export function kimiCodeModelManagerOptions( _context: OpenAICompatibleModelMapperContext<"openai-completions">, ): ModelSpec<"openai-completions"> => { const id = defaults.id; + const reasoning = kimiSupportsReasoning(entry, id); + const thinking = reasoning ? mapKimiThinking(entry) : undefined; return { ...defaults, name: typeof entry.display_name === "string" ? entry.display_name : defaults.name, - reasoning: entry.supports_reasoning === true || id.includes("thinking"), + reasoning, input: entry.supports_image_in === true || id.includes("k2.5") ? ["text", "image"] : ["text"], contextWindow: typeof entry.context_length === "number" ? entry.context_length : 262144, maxTokens: 32000, + thinking, compat: { - thinkingFormat: "zai", + thinkingFormat: thinking ? "kimi" : "zai", + kimiApiFormat: mapKimiApiFormat(entry.protocol), reasoningContentField: "reasoning_content", supportsDeveloperRole: false, }, @@ -2563,7 +2631,16 @@ function getLmStudioNativeInput(entry: Record): ("text" | "imag } function getLmStudioNativeContextWindow(entry: Record): number | undefined { + // LM Studio serves `loaded_context_length` (the window the running instance + // actually accepts) alongside `max_context_length` (the architectural + // ceiling). A model is routinely loaded below its maximum — the user picks a + // smaller window, or MLX context auto-fit shrinks it to fit unified memory — + // so prefer what the runtime serves, matching the Ollama/llama.cpp rule from + // #3754. Unloaded models report `loaded_context_length: null` and fall + // through to the max/train chain unchanged. + const loadedContextWindow = entry.state === "loaded" ? toPositiveNumber(entry.loaded_context_length, null) : null; return ( + loadedContextWindow ?? toPositiveNumber(entry.max_context_length, null) ?? toPositiveNumber(entry.context_length, null) ?? toPositiveNumber(entry.max_model_len, null) ?? @@ -2893,6 +2970,23 @@ export interface MoonshotModelManagerConfig { fetch?: FetchImpl; } +/** + * Moonshot Kimi K3 discovery metadata. K3 is dynamically discovered but absent + * from models.dev and the bundled catalog, so `mapWithBundledReference` would + * otherwise assign zero cost, null limits, text-only input, and no reasoning — + * mislabeling a paid model as "Free" (#5756). Pricing/limits from Moonshot's + * official chat-k3 pricing and quickstart guide: + * https://platform.kimi.ai/docs/pricing/chat-k3.md + * https://platform.kimi.ai/docs/guide/kimi-k3-quickstart + * K3 always reasons and supports only `reasoning_effort: "max"` — it does NOT + * use the K2.x binary `thinking: { type }` block, so the wire path routes it + * through OpenAI-style `reasoning_effort` (see `buildOpenAICompat`). + */ +const MOONSHOT_KIMI_K3_COST = { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 } as const; +const MOONSHOT_KIMI_K3_CONTEXT_WINDOW = 1_048_576; +const MOONSHOT_KIMI_K3_MAX_TOKENS = 131_072; +const MOONSHOT_KIMI_K3_THINKING: ThinkingConfig = { mode: "effort", efforts: [Effort.Max], requiresEffort: true }; + export function moonshotModelManagerOptions( config?: MoonshotModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { @@ -2915,6 +3009,25 @@ export function moonshotModelManagerOptions( const reference = references.get(defaults.id); const model = mapWithBundledReference(entry, defaults, reference); const id = model.id.toLowerCase(); + // Kimi K3 is discovered but has no bundled/models.dev reference, so the + // generic dynamic defaults would report it "Free" with no capabilities + // (#5756). Stamp the official pricing/limits when the endpoint doesn't + // carry them, and mark it reasoning + vision. K3 always reasons via + // `reasoning_effort: "max"` and does NOT use the K2.x `thinking` block, + // so its thinking config is the single-tier `max` scale — the wire path + // routes it through `reasoning_effort` (see `buildOpenAICompat`). + if (!reference && isKimiK3ModelId(id)) { + const isZeroCost = model.cost.input === 0 && model.cost.output === 0 && model.cost.cacheRead === 0; + return { + ...model, + reasoning: true, + input: ["text", "image"], + cost: isZeroCost ? { ...MOONSHOT_KIMI_K3_COST } : model.cost, + contextWindow: model.contextWindow ?? MOONSHOT_KIMI_K3_CONTEXT_WINDOW, + maxTokens: model.maxTokens ?? MOONSHOT_KIMI_K3_MAX_TOKENS, + thinking: model.thinking ?? { ...MOONSHOT_KIMI_K3_THINKING }, + }; + } // Moonshot's K2.x family (K2.5, K2.6, kimi-k2-thinking, …) is reasoning-capable // and vision-capable on the native API. Without these flags the openai-completions // path skips the z.ai-format `thinking` block, and Moonshot K2.6 stalls on first @@ -3206,17 +3319,43 @@ type LiteLLMRichEndpointModel = { hasMaxTokens: boolean; hasToolMetadata: boolean; hasSupportedOpenAIParams: boolean; + hasCost: boolean; }; +type LiteLLMRichEndpointFailure = { + endpoint: string; + reason: "http-status" | "invalid-json" | "network-error"; + status?: number; + error?: unknown; +}; +type LiteLLMRichEndpointResult = + | { models: LiteLLMRichEndpointModel[]; incompleteVisionMetadata: boolean } + | { failure: LiteLLMRichEndpointFailure }; const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const; export const OPENAI_COMPAT_DISCOVERY_DEFAULT_CONTEXT_WINDOW = 128_000; export const OPENAI_COMPAT_DISCOVERY_DEFAULT_MAX_TOKENS = 32_768; const UNKNOWN_PROXY_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } as const; +const warnedLiteLLMMetadataBases = new Set(); const LITELLM_UNUSABLE_SENTINEL_IDS: Record = { "all-team-models": true, "all-proxy-models": true, "no-default-models": true, }; +function warnLiteLLMMetadataFallback(managementBaseUrl: string, failure: LiteLLMRichEndpointFailure): void { + if (warnedLiteLLMMetadataBases.has(managementBaseUrl)) { + return; + } + warnedLiteLLMMetadataBases.add(managementBaseUrl); + logger.warn("LiteLLM rich model metadata unavailable; falling back to /v1/models", { + endpoint: `${managementBaseUrl}${failure.endpoint}`, + status: failure.status ?? "unavailable", + reason: failure.reason, + ...(failure.status === 403 + ? { requiredPermission: "Grant this LiteLLM key access to the model metadata endpoints" } + : {}), + ...(failure.error !== undefined ? { error: failure.error } : {}), + }); +} export function normalizeLiteLLMManagementBaseUrl(baseUrl: string): string { const trimmed = baseUrl.trim().replace(/\/+$/g, ""); @@ -3305,6 +3444,32 @@ function getLiteLLMMetadataValue(entry: LiteLLMRichModelEntry, key: string): unk return entry[key] ?? getLiteLLMModelInfo(entry)?.[key]; } +/** Per-million USD cost from a `*_per_token` LiteLLM field, or `undefined` when absent/non-positive. */ +function getLiteLLMPerMillionCost(entry: LiteLLMRichModelEntry, key: string): number | undefined { + const perToken = toNumber(getLiteLLMMetadataValue(entry, key)); + return perToken !== undefined && perToken > 0 ? perToken * 1_000_000 : undefined; +} + +/** + * Map LiteLLM's per-token pricing (`input_cost_per_token`, `output_cost_per_token`, + * cache costs) onto {@link ModelSpec.cost} in $/million tokens. Returns `undefined` + * when LiteLLM reports neither an input nor an output price so callers keep the + * bundled reference cost. + */ +function getLiteLLMCost(entry: LiteLLMRichModelEntry): ModelSpec["cost"] | undefined { + const input = getLiteLLMPerMillionCost(entry, "input_cost_per_token"); + const output = getLiteLLMPerMillionCost(entry, "output_cost_per_token"); + if (input === undefined && output === undefined) { + return undefined; + } + return { + input: input ?? 0, + output: output ?? 0, + cacheRead: getLiteLLMPerMillionCost(entry, "cache_read_input_token_cost") ?? 0, + cacheWrite: getLiteLLMPerMillionCost(entry, "cache_creation_input_token_cost") ?? 0, + }; +} + function getLiteLLMRichModelId(entry: LiteLLMRichModelEntry): string | undefined { return ( toNonEmptyString(entry.model_group) ?? @@ -3427,7 +3592,7 @@ function mapLiteLLMRichEntry( : (reference?.input ?? ["text"]), reasoning: typeof supportsReasoning === "boolean" ? supportsReasoning : (reference?.reasoning ?? false), thinking: reference?.thinking, - cost: reference?.cost ?? UNKNOWN_PROXY_COST, + cost: getLiteLLMCost(entry) ?? reference?.cost ?? UNKNOWN_PROXY_COST, ...(supportsTools !== undefined ? { supportsTools } : {}), compat: compat as ModelSpec["compat"], }; @@ -3439,7 +3604,7 @@ async function fetchLiteLLMRichEndpoint( managementBaseUrl: string, runtimeBaseUrl: string, signal?: AbortSignal, -): Promise<{ models: LiteLLMRichEndpointModel[]; incompleteVisionMetadata: boolean } | null> { +): Promise | null> { const fetchImpl = discoveryFetch(options.fetch); const requestHeaders: Record = { Accept: "application/json", @@ -3455,17 +3620,17 @@ async function fetchLiteLLMRichEndpoint( headers: requestHeaders, signal, }); - } catch { - return null; + } catch (error) { + return { failure: { endpoint, reason: "network-error", error } }; } if (!response.ok) { - return null; + return response.status === 404 ? null : { failure: { endpoint, reason: "http-status", status: response.status } }; } let payload: unknown; try { payload = await response.json(); - } catch { - return null; + } catch (error) { + return { failure: { endpoint, reason: "invalid-json", status: response.status, error } }; } const entries = extractLiteLLMRichEntries(payload); if (!entries || entries.length === 0) { @@ -3494,6 +3659,7 @@ async function fetchLiteLLMRichEndpoint( supportsFunctionCalling === false || supportedOpenAIParams !== undefined, hasSupportedOpenAIParams: supportedOpenAIParams !== undefined, + hasCost: getLiteLLMCost(entry) !== undefined, }); } } @@ -3516,11 +3682,27 @@ export async function fetchLiteLLMRichModels( } const fetchModels = async (signal?: AbortSignal): Promise[] | null> => { const deduped = new Map>(); + let metadataFailure: LiteLLMRichEndpointFailure | undefined; for (const endpoint of LITELLM_RICH_ENDPOINTS) { const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); if (!result) { continue; } + if ("failure" in result) { + // A 401 is a retryable auth failure owned by the caller's auth-retry + // path (withAuth refresh/sibling rotation in discoverLiteLLMModels); + // recording it here would log a fallback warning on a stale first + // credential before the refreshed retry ultimately serves rich + // metadata. Forbidden (403) and other failures are never retried, so + // they remain warn-worthy and keep priority. + if ( + result.failure.status !== 401 && + (!metadataFailure || (metadataFailure.status !== 403 && result.failure.status === 403)) + ) { + metadataFailure = result.failure; + } + continue; + } const hadPriorModels = deduped.size > 0; for (const next of result.models) { const existing = deduped.get(next.model.id); @@ -3540,6 +3722,7 @@ export async function fetchLiteLLMRichModels( ? next.model.input : existing.model.input, reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning, + cost: next.hasCost ? next.model.cost : existing.model.cost, compat: next.hasSupportedOpenAIParams ? next.model.compat : existing.model.compat, }; if (next.hasToolMetadata) { @@ -3559,6 +3742,9 @@ export async function fetchLiteLLMRichModels( } } if (deduped.size === 0) { + if (metadataFailure) { + warnLiteLLMMetadataFallback(managementBaseUrl, metadataFailure); + } return null; } return Array.from(deduped.values()) @@ -3578,13 +3764,13 @@ export function litellmModelManagerOptions( const baseUrl = config?.baseUrl ?? Bun.env.LITELLM_BASE_URL ?? "http://localhost:4000/v1"; return { providerId: "litellm", - // rich-v4 invalidates rows cached before LiteLLM ids gained bundled - // reference fallback and before discovery continued past `/model_group/info` - // when that endpoint omitted vision metadata. Earlier versions handled - // reseller usage-suffix stripping and placeholder-only `all-team-models` - // filtering; bump the version whenever the mappers below change, or warm - // authoritative caches keep serving pre-change rows for the full TTL. - cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`, + // rich-v5 invalidates rows cached before rich metadata pricing was mapped. + // Earlier versions added bundled reference fallback, continued discovery + // past incomplete `/model_group/info`, stripped reseller usage suffixes, + // and filtered placeholder-only `all-team-models` rows. Bump the version + // whenever the mappers below change, or warm authoritative caches keep + // serving pre-change rows for the full TTL. + cacheProviderId: `litellm:rich-v5:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy and is never bundled in models.json (that // would leak the machine's localhost catalog). Prefer the proxy's richer // management metadata, then enrich ids against models.dev with the bundled @@ -3724,7 +3910,8 @@ export interface GithubCopilotModelManagerConfig { const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus|fable|mythos)-\d/; const isCopilotResponsesModelId = (modelId: string): boolean => - modelId.startsWith("gpt-5") || modelId.startsWith("oswe"); + modelId.startsWith("gpt-5") || modelId.startsWith("oswe") || modelId.startsWith("mai-"); +const COPILOT_CACHE_INVALIDATED_MODEL_IDS = ["mai-code-1-flash-picker"]; function inferCopilotApi(modelId: string): Api { if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) { @@ -3888,6 +4075,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana const resolveReference = createReferenceResolver(providerRefs); return { providerId: "github-copilot", + dropCachedModelIdsOnStaticMismatch: COPILOT_CACHE_INVALIDATED_MODEL_IDS, ...(apiKey && { fetchDynamicModels: async () => { const longContextVariants: ModelSpec[] = []; @@ -4505,9 +4693,17 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDescriptor[] = [ // --- zAI --- - anthropicMessagesDescriptor("zai-coding-plan", "zai", "https://api.z.ai/api/anthropic"), - // --- Umans AI Coding Plan --- - anthropicMessagesDescriptor("umans-ai-coding-plan", "umans", UMANS_BASE_URL), + // Source the models.dev `zai` (pay-as-you-go) key rather than `zai-coding-plan`: + // the coding-plan key reports all-$0 subscription rates, which surface every GLM + // SKU as "Free" in `/models`. The PAYG key carries the real per-token rates for + // the identical model ids, so the enumerated token costs line up with the other + // subscription providers for comparison (issue #5598). + anthropicMessagesDescriptor("zai", "zai", "https://api.z.ai/api/anthropic"), + // --- Umans AI --- + // Source the pay-as-you-go catalog: the coding-plan key publishes subscription + // costs as zero, while `/models/info` omits pricing entirely. The generator + // overlays these rates onto the authoritative endpoint discovery (issue #5733). + anthropicMessagesDescriptor("umans-ai", "umans", UMANS_BASE_URL), // --- Xiaomi --- openAiCompletionsDescriptor("xiaomi", "xiaomi", "https://api.xiaomimimo.com/v1", { defaultContextWindow: 262144, diff --git a/packages/catalog/src/provider-models/special.ts b/packages/catalog/src/provider-models/special.ts index 7a18d0069..d9fd2a70a 100644 --- a/packages/catalog/src/provider-models/special.ts +++ b/packages/catalog/src/provider-models/special.ts @@ -1,37 +1,89 @@ import { once } from "@oh-my-pi/pi-utils"; -import { fetchCodexModels } from "../discovery/codex"; +import { type CodexModelDiscoveryResult, fetchCodexModels } from "../discovery/codex"; import type { DevinModelDiscoveryOptions } from "../discovery/devin"; import { buildGitLabDuoWorkflowFallbackModel, fetchGitLabDuoWorkflowModels } from "../discovery/gitlab-duo-workflow"; import type { ModelManagerOptions } from "../model-manager"; -import type { FetchImpl } from "../types"; +import type { FetchImpl, ModelSpec } from "../types"; // --------------------------------------------------------------------------- // OpenAI Codex // --------------------------------------------------------------------------- -export interface OpenAICodexModelManagerConfig { - accessToken?: string; +/** One Codex OAuth account to fetch a catalog for. */ +export interface OpenAICodexAccount { + /** OAuth access token used for `Authorization: Bearer ...`. */ + accessToken: string; + /** ChatGPT account id sent as the `chatgpt-account-id` header. */ accountId?: string; +} + +export interface OpenAICodexModelManagerConfig { + /** + * Resolves every configured Codex OAuth account at discovery time. Codex + * discovery is account-scoped — a model can be available to one account and + * absent from another — so each account's `/models` endpoint is fetched + * independently and the results unioned by id. Without this, discovery would + * surface only the account it happened to resolve and, being authoritative, + * prune every model the other accounts expose (#6265). + * + * Returns `null` to abort discovery entirely (e.g. an account's credential + * failed to refresh): a partial account set would be cached as the complete + * authoritative catalog and hide the missing account's models, so the caller + * keeps the previous/bundled catalog instead. + */ + resolveAccounts?: () => Promise; clientVersion?: string; + fetch?: FetchImpl; } export function openaiCodexModelManagerOptions( config: OpenAICodexModelManagerConfig = {}, ): ModelManagerOptions<"openai-codex-responses"> { - const { accessToken, accountId, clientVersion } = config; + const { resolveAccounts, clientVersion, fetch } = config; return { providerId: "openai-codex", - ...(accessToken + dynamicModelsAuthoritative: true, + ...(resolveAccounts ? { fetchDynamicModels: async () => { - const result = await fetchCodexModels({ accessToken, accountId, clientVersion }); - return result?.models ?? null; + const accounts = await resolveAccounts(); + if (!accounts || accounts.length === 0) return null; + const results = await Promise.all( + accounts.map(account => + fetchCodexModels({ + accessToken: account.accessToken, + accountId: account.accountId, + clientVersion, + fetchFn: fetch, + }), + ), + ); + return unionCodexModels(results); }, } : undefined), }; } +/** + * Merge complete per-account Codex catalogs into one authoritative list, + * deduped by model id (first account to expose an id wins). Returns `null` when + * any account's fetch failed, so a partial list cannot replace the previous or + * bundled authoritative catalog. + */ +function unionCodexModels( + results: readonly (CodexModelDiscoveryResult | null)[], +): ModelSpec<"openai-codex-responses">[] | null { + const byId = new Map>(); + for (const result of results) { + if (!result) return null; + for (const model of result.models) { + if (!byId.has(model.id)) byId.set(model.id, model); + } + } + return [...byId.values()]; +} + // --------------------------------------------------------------------------- // Cursor // --------------------------------------------------------------------------- diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index ec5d93526..143fd4565 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -148,7 +148,7 @@ export interface Usage { }; } -export type OpenAIReasoningFormat = "openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template"; +export type OpenAIReasoningFormat = "openai" | "openrouter" | "zai" | "kimi" | "qwen" | "qwen-chat-template"; export type OpenAIReasoningDisableMode = | "omit" @@ -205,8 +205,10 @@ export interface OpenAICompat { requiresThinkingAsText?: boolean; /** Whether tool call IDs must be normalized to Mistral format (exactly 9 alphanumeric chars). Default: auto-detected from URL. */ requiresMistralToolIds?: boolean; - /** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "openrouter" uses reasoning: { effort }, "zai" uses thinking: { type: "enabled" | "disabled" } (also used by Moonshot Kimi), "qwen" uses top-level enable_thinking, and "qwen-chat-template" uses chat_template_kwargs.enable_thinking. Default: "openai". */ + /** Format for reasoning/thinking parameter. `"kimi"` uses `thinking: { type, effort }`; other values select their provider-native reasoning fields. Default: `"openai"`. */ thinkingFormat?: OpenAIReasoningFormat; + /** Kimi Code transport selected by live per-model protocol metadata. User settings take precedence. */ + kimiApiFormat?: "openai" | "anthropic"; /** Request-time disable encoding for the selected reasoning/thinking format. Default: derived from `thinkingFormat`. */ reasoningDisableMode?: OpenAIReasoningDisableMode; /** Whether the provider rejects `reasoning.effort`/`reasoning_effort` even when the model reasons natively. Default: false unless reasoning effort is unsupported. */ @@ -308,14 +310,29 @@ export interface OpenAICompat { supportsStrictMode?: boolean; /** * Tool-schema dialect the endpoint validates `tools.function.parameters` - * against. `"moonshot-mfjs"` triggers Moonshot Flavored JSON Schema - * normalization (collapse `const`→`enum`, infer `type` on bare enums, strip - * unsupported validators/`prefixItems`) because Moonshot/Kimi native hosts - * reject standard JSON Schema constructs with HTTP 400. Default: - * auto-detected (`"moonshot-mfjs"` on api.moonshot.ai / api.kimi.com). Set - * `"none"` to opt a custom Moonshot-compatible host out. + * against. + * + * `"moonshot-mfjs"` triggers Moonshot Flavored JSON Schema normalization + * (collapse `const`→`enum`, infer `type` on bare enums, strip unsupported + * validators/`prefixItems`) because Moonshot/Kimi native hosts reject + * standard JSON Schema constructs with HTTP 400. + * + * `"grammar"` triggers grammar-sampler normalization (widen bare boolean + * `true`/`{}` subschemas in genuine subschema slots into a value-accepting + * union of primitives) because grammar-constrained backends (llama.cpp, LM + * Studio, vLLM) build a GBNF grammar from the JSON Schema and 400 with + * `Unrecognized schema: true` on a bare boolean subschema (issue #5914). + * Boolean `additionalProperties`/`unevaluatedProperties` are preserved — the + * grammar converter reads them as closed/open-object semantics, and + * `additionalProperties: false` pins the strict object shape. + * + * Default: auto-detected — `"moonshot-mfjs"` on Moonshot native hosts + * (api.moonshot.ai / api.kimi.com) and Kimi-family model ids on any host, + * since proxies (OpenRouter, custom gateways) forward schemas to Moonshot + * verbatim; `"grammar"` on local OpenAI-compatible backends. Set `"none"` + * to opt a host out. */ - toolSchemaFlavor?: "moonshot-mfjs" | "none"; + toolSchemaFlavor?: "moonshot-mfjs" | "grammar" | "none"; /** * Stream-watchdog idle-timeout floor in ms for slow reasoning hosts. * Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning). @@ -327,6 +344,15 @@ export interface OpenAICompat { toolStrictMode?: "all_strict" | "none"; /** Whether request shaping may send reasoning params at all. Default: auto-detected (disabled for GitHub Copilot chat-completions). */ supportsReasoningParams?: boolean; + /** + * Whether the endpoint accepts explicit sampling parameters (`temperature`, + * `top_p`, `top_k`, `min_p`, penalties). OpenAI proprietary reasoning models + * (o-series, gpt-5+) reject them with `400 Unsupported parameter: + * 'temperature' is not supported with this model` on every serving host + * (official, Azure, GitHub Copilot). When unset, auto-detected from the + * model id. Default: true. Issue #5606. + */ + supportsSamplingParams?: boolean; /** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */ alwaysSendMaxTokens?: boolean; /** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */ @@ -372,7 +398,7 @@ export interface AnthropicCompat { * tags: 'disabled', 'enabled'`. */ disableAdaptiveThinking?: boolean; - /** Whether tools may include Anthropic's per-tool eager_input_streaming flag. Default: true. */ + /** Whether tools may include Anthropic's per-tool eager_input_streaming flag. Default: true for the canonical Anthropic API. */ supportsEagerToolInputStreaming?: boolean; /** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */ supportsLongCacheRetention?: boolean; @@ -405,6 +431,11 @@ export interface AnthropicCompat { * auto-detected (Z.AI hosts). */ requiresToolResultId?: boolean; + /** + * Allow configured Claude Code fingerprint headers to replace generated + * OAuth defaults on non-official Anthropic endpoints. + */ + allowAnthropicHeaderOverrides?: boolean; /** * Replay unsigned `thinking` blocks from prior assistant turns as native * thinking instead of demoting them to text. Official Anthropic enforces @@ -464,7 +495,10 @@ export interface ResolvedOpenAISharedCompat { supportsReasoningEffort: boolean; reasoningEffortMap: Partial>; supportsReasoningParams: boolean; + supportsSamplingParams: boolean; thinkingFormat: OpenAIReasoningFormat; + /** Kimi Code transport selected by live per-model protocol metadata. */ + kimiApiFormat?: OpenAICompat["kimiApiFormat"]; reasoningDisableMode: OpenAIReasoningDisableMode; omitReasoningEffort: boolean; includeEncryptedReasoning: boolean; @@ -500,6 +534,8 @@ export interface ResolvedOpenAISharedCompat { openRouterRouting?: OpenAICompat["openRouterRouting"]; /** Provider-specific wire model-id transform applied to the base id. */ wireModelIdMode: "raw" | "firepass" | "fireworks" | "openrouter"; + /** See {@link OpenAICompat.toolSchemaFlavor}. Read by both wire paths when converting tools. */ + toolSchemaFlavor?: OpenAICompat["toolSchemaFlavor"]; } /** @@ -516,7 +552,9 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & | "supportsReasoningEffort" | "reasoningEffortMap" | "supportsReasoningParams" + | "supportsSamplingParams" | "thinkingFormat" + | "kimiApiFormat" | "reasoningDisableMode" | "omitReasoningEffort" | "includeEncryptedReasoning" @@ -568,7 +606,6 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & thinkingKeep?: OpenAICompat["thinkingKeep"]; streamIdleTimeoutMs?: number; toolStrictMode: ResolvedToolStrictMode; - toolSchemaFlavor?: OpenAICompat["toolSchemaFlavor"]; /** The model sits behind Vercel AI Gateway. */ isVercelGatewayHost: boolean; dropThinkingWhenReasoningEffort: boolean; diff --git a/packages/catalog/src/variant-collapse.ts b/packages/catalog/src/variant-collapse.ts index 770d7610d..e7b8e571f 100644 --- a/packages/catalog/src/variant-collapse.ts +++ b/packages/catalog/src/variant-collapse.ts @@ -227,13 +227,12 @@ const GEMINI_3_PRO_FAMILY_BUDGETS: Readonly>> = { }; /** - * The two Cloud Code Assist providers share the same Antigravity discovery list - * but disagree on the thinking transport: `google-antigravity` (daily-cloudcode-pa) - * sends an explicit `thinkingBudget` (verified against captured requests), while - * `google-gemini-cli` (cloudcode-pa) follows the official Gemini CLI and uses - * `thinkingLevel`. The Gemini 3.x families therefore differ only in thinking - * transport (and, for Flash, the per-tier wire-id routing); everything else is - * shared verbatim. + * Cloud Code Assist's legacy Gemini 3.5 Flash and 3.1 Pro families use + * different thinking transports: `google-antigravity` (daily-cloudcode-pa) + * sends captured `thinkingBudget` values, while `google-gemini-cli` + * (cloudcode-pa) follows the official Gemini CLI and uses `thinkingLevel`. + * Gemini 3.6 exposes one wire id per level and uses `thinkingLevel` on both + * endpoints. */ function geminiFlashFamily(mode: "budget" | "google-level"): EffortVariantFamily { const budget = mode === "budget"; @@ -266,6 +265,23 @@ function geminiFlashFamily(mode: "budget" | "google-level"): EffortVariantFamily }; } +const GEMINI_36_FLASH_FAMILY: EffortVariantFamily = { + id: "gemini-3.6-flash", + name: "Gemini 3.6 Flash", + members: ["gemini-3.6-flash-low", "gemini-3.6-flash-medium", "gemini-3.6-flash-high", "gemini-3.6-flash-tiered"], + routing: { + [Effort.Minimal]: "gemini-3.6-flash-low", + [Effort.Low]: "gemini-3.6-flash-low", + [Effort.Medium]: "gemini-3.6-flash-medium", + [Effort.High]: "gemini-3.6-flash-high", + }, + thinking: { + mode: "google-level", + efforts: GEMINI_3_FLASH_FAMILY_EFFORTS, + requiresEffort: true, + }, +}; + function geminiProFamily(mode: "budget" | "google-level"): EffortVariantFamily { const budget = mode === "budget"; return { @@ -346,14 +362,19 @@ const SHARED_CCA_FAMILIES: readonly EffortVariantFamily[] = [ thinkingPair("gemini-2.5-flash", "Gemini 2.5 Flash"), ]; -/** `google-antigravity` (daily-cloudcode-pa): Gemini 3.x on the budget transport. */ +/** `google-antigravity` Gemini families, using each generation's native transport. */ export const ANTIGRAVITY_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { - families: [geminiFlashFamily("budget"), geminiProFamily("budget"), ...SHARED_CCA_FAMILIES], + families: [GEMINI_36_FLASH_FAMILY, geminiFlashFamily("budget"), geminiProFamily("budget"), ...SHARED_CCA_FAMILIES], }; -/** `google-gemini-cli` (cloudcode-pa): Gemini 3.x on the level transport (official CLI parity). */ +/** `google-gemini-cli` Gemini families on the official CLI's level transport. */ export const GEMINI_CLI_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { - families: [geminiFlashFamily("google-level"), geminiProFamily("google-level"), ...SHARED_CCA_FAMILIES], + families: [ + GEMINI_36_FLASH_FAMILY, + geminiFlashFamily("google-level"), + geminiProFamily("google-level"), + ...SHARED_CCA_FAMILIES, + ], }; export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { families: [ @@ -532,6 +553,39 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High], ), + // GLM-5.2 200K — only the base wire UID `glm-5-2` is free on Devin's + // Coding Plan (verified via streamDevin: `glm-5-2-none` and `glm-5-2-max` + // both return "weekly usage quota exhausted" while `glm-5-2` streams + // successfully). Route every effort to `glm-5-2` so the collapsed entry + // is always free; include the paid 200K variants as members so they are + // hidden from the model list. The 1M-context variants stay as separate + // paid entries (collapsed below). + { + id: "glm-5-2", + name: "GLM-5.2", + members: ["glm-5-2", "glm-5-2-none", "glm-5-2-max"], + routing: { + [Effort.High]: "glm-5-2", + [Effort.XHigh]: "glm-5-2", + }, + thinking: { + mode: "effort", + efforts: [Effort.High, Effort.XHigh], + requiresEffort: true, + }, + }, + // GLM-5.2 1M — paid variants that consume weekly quota. Collapse the + // three 1M-context variants into one entry with proper effort routing. + devinTierFamily( + "glm-5-2-1m", + "GLM-5.2 1M", + { + off: "glm-5-2-none-1m", + high: "glm-5-2-1m", + xhigh: "glm-5-2-max-1m", + }, + [Effort.High, Effort.XHigh], + ), ], }; diff --git a/packages/catalog/test/build.test.ts b/packages/catalog/test/build.test.ts index bbde3abc9..0524f933d 100644 --- a/packages/catalog/test/build.test.ts +++ b/packages/catalog/test/build.test.ts @@ -282,6 +282,31 @@ describe("openai-completions wire-quirk compat detection", () => { expect(buildOpenAICompat(completionsSpec()).reasoningDeltasMayBeCumulative).toBe(false); }); + it("extends the reasoning stream idle floor to Kimi K2.6 and K2.7 Code, not other reasoning models", () => { + const kimiOverrides = { + provider: "moonshot", + baseUrl: "https://api.moonshot.ai/v1", + reasoning: true, + } as const; + expect(buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.6" })).streamIdleTimeoutMs).toBe( + 300_000, + ); + expect(buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.7-code" })).streamIdleTimeoutMs).toBe( + 300_000, + ); + expect( + buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.7-code-highspeed" })).streamIdleTimeoutMs, + ).toBe(300_000); + // K2.7 Code on non-native OpenAI-compatible hosts keeps their default. + expect( + buildOpenAICompat(completionsSpec({ id: "kimi-k2.7-code", reasoning: true })).streamIdleTimeoutMs, + ).toBeUndefined(); + // A non-Kimi reasoning model on a generic host keeps the runtime default. + expect( + buildOpenAICompat(completionsSpec({ id: "some-reasoner", reasoning: true })).streamIdleTimeoutMs, + ).toBeUndefined(); + }); + it("maps the remaining provider-keyed wire quirks", () => { expect(buildOpenAICompat(completionsSpec({ provider: "ollama" })).emptyLengthFinishIsContextError).toBe(true); expect(buildOpenAICompat(completionsSpec()).emptyLengthFinishIsContextError).toBe(false); @@ -297,6 +322,36 @@ describe("openai-completions wire-quirk compat detection", () => { expect(buildOpenAICompat(completionsSpec()).dropThinkingWhenReasoningEffort).toBe(false); }); + it("floors the stream timeout for a loopback litellm proxy without enabling reasoning replay (#4786)", () => { + // A litellm proxy on a loopback baseUrl fronts a local llama-server whose + // prefill can exceed the 100s default first-event budget on large prompts. + // The proxy carve-out (which keeps `replayReasoningContent` off so the + // field is never forwarded to an unrelated cloud upstream) must NOT also + // strip the widened stream-timeout floor, or the turn aborts and + // retry-loops during a slow reprocess. + const loopback = buildOpenAICompat( + completionsSpec({ provider: "litellm", id: "qwen3", baseUrl: "http://127.0.0.1:4000/v1" }), + ); + expect(loopback.streamIdleTimeoutMs).toBe(300_000); + expect(loopback.replayReasoningContent).toBe(false); + + // A litellm proxy on a remote baseUrl gets neither: no local upstream to + // wait on, and replay would risk a 400 on the cloud upstream. + const remote = buildOpenAICompat( + completionsSpec({ provider: "litellm", id: "qwen3", baseUrl: "https://litellm.example.com/v1" }), + ); + expect(remote.streamIdleTimeoutMs).toBeUndefined(); + expect(remote.replayReasoningContent).toBe(false); + + // A first-party local backend (llama.cpp) still gets both the floor and + // the reasoning replay it needs for KV-cache reuse. + const native = buildOpenAICompat( + completionsSpec({ provider: "llama.cpp", id: "qwen3", baseUrl: "http://127.0.0.1:8080/v1" }), + ); + expect(native.streamIdleTimeoutMs).toBe(300_000); + expect(native.replayReasoningContent).toBe(true); + }); + it("disables the leaked-markup healer for the official OpenAI endpoint only", () => { // Official OpenAI returns structured reasoning and never leaks fences, so // the provider-local healer stays off; every other OpenAI-compatible host @@ -538,6 +593,193 @@ describe("model cache spec round trip", () => { await fs.rm(tempDir, { recursive: true, force: true }); } }); + it("restores static model headers on fresh cache reads", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-static-headers-")); + const dbPath = path.join(tempDir, "models.db"); + const staticModel = completionsSpec({ + id: "header-static-model", + provider: "header-cache-test", + headers: { "X-Project-Id": "project-42" }, + }); + let fetches = 0; + const options = { + providerId: "header-cache-test", + staticModels: [staticModel], + cacheDbPath: dbPath, + fetchDynamicModels: async () => { + fetches++; + return []; + }, + }; + try { + const online = await resolveProviderModels(options, "online"); + expect(online.models[0]?.headers).toEqual({ "X-Project-Id": "project-42" }); + expect(fetches).toBe(1); + + const offline = await resolveProviderModels(options, "offline"); + expect(offline.models[0]?.headers).toEqual({ "X-Project-Id": "project-42" }); + expect(fetches).toBe(1); + + const fresh = await resolveProviderModels(options, "online-if-uncached"); + expect(fresh.models[0]?.headers).toEqual({ "X-Project-Id": "project-42" }); + expect(fetches).toBe(1); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("refetches dynamic-only models whose headers cannot be restored", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-dynamic-headers-")); + const dbPath = path.join(tempDir, "models.db"); + const dynamicModel = completionsSpec({ + id: "header-dynamic-model", + provider: "header-cache-test", + headers: { "X-Required-Route": "route-42" }, + }); + let fetches = 0; + const options = { + providerId: "header-cache-test", + staticModels: [], + dynamicModelsAuthoritative: true, + cacheDbPath: dbPath, + fetchDynamicModels: async () => { + fetches++; + return [dynamicModel]; + }, + }; + try { + const online = await resolveProviderModels(options, "online"); + expect(online.models[0]?.headers).toEqual({ "X-Required-Route": "route-42" }); + expect(fetches).toBe(1); + + const fresh = await resolveProviderModels(options, "online-if-uncached"); + expect(fresh.models[0]?.headers).toEqual({ "X-Required-Route": "route-42" }); + expect(fetches).toBe(2); + + const offline = await resolveProviderModels(options, "offline"); + expect(offline.models).toEqual([]); + expect(offline.stale).toBe(true); + expect(fetches).toBe(2); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("keeps a synthesized request-model variant across an offline restart", async () => { + // Regression for #6037/#6284: Copilot `-1m` long-context variants are + // synthesized dynamically with transport headers and a `requestModelId` + // pointing at a same-provider base. Their headers are omitted from the + // cache but recoverable from the base's static headers, so they must NOT + // be flagged unrestorable and dropped on the next offline read. + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-request-model-variant-")); + const dbPath = path.join(tempDir, "models.db"); + const headers = { "X-GitHub-Api-Version": "2026-06-01" }; + const base = completionsSpec({ id: "sol", provider: "variant-cache-test", headers }); + const variant = completionsSpec({ + id: "sol-1m", + provider: "variant-cache-test", + requestModelId: "sol", + headers, + contextWindow: 1_000_000, + }); + const options = { + providerId: "variant-cache-test", + staticModels: [base], + cacheDbPath: dbPath, + }; + try { + const online = await resolveProviderModels<"openai-completions">( + { ...options, fetchDynamicModels: async () => [base, variant] }, + "online", + ); + expect(online.models.find(candidate => candidate.id === "sol-1m")).toBeDefined(); + + const offline = await resolveProviderModels<"openai-completions">( + { ...options, fetchDynamicModels: async () => null }, + "offline", + ); + const restored = offline.models.find(candidate => candidate.id === "sol-1m"); + expect(restored).toBeDefined(); + expect(restored?.headers).toEqual(headers); + expect(offline.models.find(candidate => candidate.id === "sol")?.headers).toEqual(headers); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("refetches a current request-model alias whose headers differ from its static base", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-custom-alias-")); + const dbPath = path.join(tempDir, "models.db"); + const baseHeaders = { "X-Route": "static" }; + const customHeaders = { "X-Route": "tenant-specific" }; + const base = completionsSpec({ id: "base", provider: "alias-cache-test", headers: baseHeaders }); + const aliasSpec = completionsSpec({ + id: "custom-alias", + provider: "alias-cache-test", + requestModelId: "base", + headers: customHeaders, + }); + const alias = buildModel(aliasSpec); + let fetches = 0; + const options = { + providerId: "alias-cache-test", + staticModels: [base], + cacheDbPath: dbPath, + fetchDynamicModels: async () => { + fetches++; + return [aliasSpec]; + }, + }; + try { + writeModelCache("alias-cache-test", Date.now(), [alias], true, "", dbPath, [buildModel(base)]); + + const refreshed = await resolveProviderModels<"openai-completions">(options, "online-if-uncached"); + expect(fetches).toBe(1); + expect(refreshed.models.find(candidate => candidate.id === alias.id)?.headers).toEqual(customHeaders); + + const offline = await resolveProviderModels<"openai-completions">(options, "offline"); + expect(offline.models.find(candidate => candidate.id === alias.id)).toBeUndefined(); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("recovers a legacy stale-marked request-model variant via requestModelId", async () => { + // Legacy cache rows (written by the old id-only writer) flag `-1m` + // variants unrestorable because it never matched their base's headers. + // The restore path must still recover them through `requestModelId`. + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-legacy-variant-")); + const dbPath = path.join(tempDir, "models.db"); + const headers = { "X-GitHub-Api-Version": "2026-06-01" }; + const base = completionsSpec({ id: "sol", provider: "variant-cache-test", headers }); + const variant = buildModel( + completionsSpec({ + id: "sol-1m", + provider: "variant-cache-test", + requestModelId: "sol", + headers, + contextWindow: 1_000_000, + }), + ); + try { + // Emulate a legacy write: no static header source, so the variant is + // flagged unrestorable even though its base carries the headers. + writeModelCache("variant-cache-test", Date.now(), [variant], true, "", dbPath); + const db = new Database(dbPath); + db.run("UPDATE model_cache SET header_restore_version = 0 WHERE provider_id = ?", ["variant-cache-test"]); + db.close(); + + const offline = await resolveProviderModels<"openai-completions">( + { providerId: "variant-cache-test", staticModels: [base], cacheDbPath: dbPath }, + "offline", + ); + const restored = offline.models.find(candidate => candidate.id === "sol-1m"); + expect(restored).toBeDefined(); + expect(restored?.headers).toEqual(headers); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); }); describe("isOfficialAnthropicApiUrl", () => { diff --git a/packages/catalog/test/codex-discovery.test.ts b/packages/catalog/test/codex-discovery.test.ts index b647ac895..0c6a94791 100644 --- a/packages/catalog/test/codex-discovery.test.ts +++ b/packages/catalog/test/codex-discovery.test.ts @@ -7,6 +7,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { fetchCodexModels } from "@oh-my-pi/pi-catalog/discovery/codex"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager"; +import { openaiCodexModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/special"; import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; describe("Codex model discovery", () => { @@ -100,6 +101,262 @@ describe("Codex model discovery", () => { expect(legacy?.useResponsesLite).toBeUndefined(); }); + it("falls back to the 372K window for GPT-5.6 SKUs when upstream omits context_window (#5705)", async () => { + const fetchFn: typeof fetch = Object.assign( + async () => + new Response( + JSON.stringify({ + models: [ + { + slug: "gpt-5.6-sol", + display_name: "GPT-5.6-Sol", + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + }, + { + slug: "gpt-5.5", + display_name: "GPT-5.5", + default_reasoning_level: "high", + supported_reasoning_levels: ["low", "high"], + input_modalities: ["text"], + supported_in_api: true, + }, + ], + }), + ), + { preconnect() {} }, + ); + const result = await fetchCodexModels({ + accessToken: "test-token", + baseUrl: "https://codex.example/backend-api", + clientVersion: "0.99.0", + fetchFn, + }); + + const sol = result?.models.find(model => model.id === "gpt-5.6-sol"); + expect(sol?.contextWindow).toBe(372_000); + const legacy = result?.models.find(model => model.id === "gpt-5.5"); + expect(legacy?.contextWindow).toBe(272_000); + }); + + it("floors GPT-5.6 SKUs to 372K when upstream actively reports 272000 (#6259)", async () => { + const fetchFn: typeof fetch = Object.assign( + async () => + new Response( + JSON.stringify({ + models: [ + { + slug: "gpt-5.6-sol", + display_name: "GPT-5.6-Sol", + context_window: 272_000, + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + }, + { + slug: "gpt-5.5", + display_name: "GPT-5.5", + context_window: 272_000, + default_reasoning_level: "high", + supported_reasoning_levels: ["low", "high"], + input_modalities: ["text"], + supported_in_api: true, + }, + ], + }), + ), + { preconnect() {} }, + ); + const result = await fetchCodexModels({ + accessToken: "test-token", + baseUrl: "https://codex.example/backend-api", + clientVersion: "0.144.1", + fetchFn, + }); + + const sol = result?.models.find(model => model.id === "gpt-5.6-sol"); + expect(sol?.contextWindow).toBe(372_000); + // Non-5.6 SKUs still honor the reported value verbatim. + const legacy = result?.models.find(model => model.id === "gpt-5.5"); + expect(legacy?.contextWindow).toBe(272_000); + }); + + it("keeps account-listed API-unsupported models while pruning hidden and absent models", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-authoritative-")); + const staticOnlyModel: ModelSpec<"openai-codex-responses"> = { + id: "unsupported-static", + name: "Unsupported static model", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.com/backend-api", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 272_000, + maxTokens: 128_000, + }; + const sparkModel: ModelSpec<"openai-codex-responses"> = { + ...staticOnlyModel, + id: "gpt-5.3-codex-spark", + name: "GPT-5.3 Codex Spark", + contextWindow: 128_000, + }; + const fetchFn: typeof fetch = Object.assign( + async () => + new Response( + JSON.stringify({ + models: [ + { + slug: "gpt-5.3-codex-spark", + display_name: "GPT-5.3-Codex-Spark", + visibility: "list", + supported_in_api: false, + context_window: 128_000, + default_reasoning_level: "high", + input_modalities: ["text"], + }, + { + slug: "hidden-model", + display_name: "Hidden model", + visibility: "hidden", + supported_in_api: true, + }, + { + slug: "hide-model", + display_name: "Hide model", + visibility: "hide", + supported_in_api: true, + }, + ], + }), + ), + { preconnect() {} }, + ); + try { + const result = await resolveProviderModels( + { + ...openaiCodexModelManagerOptions({ + resolveAccounts: async () => [{ accessToken: "test-token" }], + fetch: fetchFn, + }), + staticModels: [staticOnlyModel, sparkModel], + cacheDbPath: path.join(tempDir, "models.db"), + }, + "online", + ); + + expect(result.models.map(model => model.id)).toEqual(["gpt-5.3-codex-spark"]); + expect(result.models[0]).toMatchObject({ + contextWindow: 128_000, + maxTokens: 128_000, + }); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("unions models across every configured Codex OAuth account (#6265)", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-union-")); + // Codex `/models` is account-scoped: account 1 lacks gpt-5.6-sol, account 2 + // exposes it. Keyed off the chatgpt-account-id header the discovery flow + // sends per account. + const catalogs: Record = { + "account-1": ["gpt-5.6-terra", "gpt-5.6-luna"], + "account-2": ["gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"], + }; + const fetchFn: typeof fetch = Object.assign( + async (_input: string | URL | Request, init?: RequestInit) => { + const accountId = new Headers(init?.headers).get("chatgpt-account-id") ?? ""; + const slugs = catalogs[accountId] ?? []; + return new Response( + JSON.stringify({ + models: slugs.map(slug => ({ + slug, + display_name: slug, + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + })), + }), + ); + }, + { preconnect() {} }, + ); + try { + const options = openaiCodexModelManagerOptions({ + resolveAccounts: async () => [ + { accessToken: "token-1", accountId: "account-1" }, + { accessToken: "token-2", accountId: "account-2" }, + ], + fetch: fetchFn, + }); + const result = await resolveProviderModels( + { ...options, cacheDbPath: path.join(tempDir, "models.db") }, + "online", + ); + + expect(result.models.map(model => model.id).sort()).toEqual(["gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + + it("keeps bundled Codex models when any account catalog fetch fails (#6265)", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-union-fail-")); + const bundled: ModelSpec<"openai-codex-responses"> = { + id: "gpt-5.6-terra", + name: "GPT-5.6 Terra", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.com/backend-api", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 372_000, + maxTokens: 128_000, + }; + const fetchFn: typeof fetch = Object.assign( + async (_input: string | URL | Request, init?: RequestInit) => { + const accountId = new Headers(init?.headers).get("chatgpt-account-id"); + if (accountId === "account-1") { + return Response.json({ + models: [ + { + slug: "partial-account-model", + display_name: "Partial Account Model", + supported_in_api: true, + input_modalities: ["text"], + }, + ], + }); + } + return new Response("nope", { status: 500 }); + }, + { preconnect() {} }, + ); + try { + const options = openaiCodexModelManagerOptions({ + resolveAccounts: async () => [ + { accessToken: "token-1", accountId: "account-1" }, + { accessToken: "token-2", accountId: "account-2" }, + ], + fetch: fetchFn, + }); + const result = await resolveProviderModels( + { ...options, staticModels: [bundled], cacheDbPath: path.join(tempDir, "models.db") }, + "online", + ); + + expect(result.models.map(model => model.id)).toEqual(["gpt-5.6-terra"]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("ignores pre-V2 Codex discovery cache rows", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-v7-cache-")); const dbPath = path.join(tempDir, "models.db"); diff --git a/packages/catalog/test/fireworks-serverless-discovery.test.ts b/packages/catalog/test/fireworks-serverless-discovery.test.ts index 9fc89b93c..834d70833 100644 --- a/packages/catalog/test/fireworks-serverless-discovery.test.ts +++ b/packages/catalog/test/fireworks-serverless-discovery.test.ts @@ -10,6 +10,7 @@ */ import { describe, expect, it } from "bun:test"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { fireworksModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; @@ -132,8 +133,10 @@ describe("Fireworks control-plane serverless discovery", () => { expect(kimi.provider).toBe("fireworks"); expect(kimi.baseUrl).toBe("https://api.fireworks.ai/inference/v1"); expect(kimi.contextWindow).toBe(262144); - // Kimi family clamps to the published 32,768 output cap. - expect(kimi.maxTokens).toBe(32768); + // K2.7-Code is excluded from the K2.5/K2.6 32,768 cap; its ceiling stays + // in lockstep with the bundled reference rather than a pinned constant. + expect(kimi.maxTokens).toBe(getBundledModel("fireworks", "kimi-k2.7-code")?.maxTokens ?? null); + expect(kimi.maxTokens).toBeGreaterThan(32_768); expect(kimi.input).toEqual(["text", "image"]); // Control plane reports no reasoning bit; serverless chat LLMs default on. expect(kimi.reasoning).toBe(true); diff --git a/packages/catalog/test/gateway-reference.test.ts b/packages/catalog/test/gateway-reference.test.ts new file mode 100644 index 000000000..227cd5a65 --- /dev/null +++ b/packages/catalog/test/gateway-reference.test.ts @@ -0,0 +1,18 @@ +import { describe, expect, test } from "bun:test"; +import { getBundledModelReferenceIndex } from "../src/identity/bundled"; +import { inheritReferenceThinking, resolveModelReference } from "../src/identity/reference"; + +describe("Portkey gateway model references", () => { + test("@modal ids do not fuzzy-match bundled catalog entries", () => { + const index = getBundledModelReferenceIndex(); + expect(resolveModelReference("@modal/GLM-5-2-FP8", index)).toBeUndefined(); + }); + + test("cross-provider references do not inherit wire routing thinking", () => { + const index = getBundledModelReferenceIndex(); + const kiloGigaPotato = resolveModelReference("giga-potato", index); + expect(kiloGigaPotato?.provider).toBe("kilo"); + expect(kiloGigaPotato?.thinking?.effortRouting).toBeDefined(); + expect(inheritReferenceThinking(undefined, kiloGigaPotato, "gateway")).toBeUndefined(); + }); +}); diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index 2c6da8a51..151b22409 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -86,6 +86,39 @@ describe("generated model policies", () => { expect(models[3]?.priority).toBe(1); }); + it("pins GPT-5.6 Codex-transport context window to the 372K hard capacity (#5705)", () => { + const models: ModelSpec[] = [ + // Codex discovery underreports these via DEFAULT_CONTEXT_WINDOW=272000. + createSpec({ + id: "gpt-5.6-luna", + api: "openai-codex-responses", + provider: "openai-codex", + contextWindow: 272000, + }), + createSpec({ + id: "gpt-5.6-sol", + api: "openai-codex-responses", + provider: "openai-codex", + contextWindow: 272000, + }), + createSpec({ + id: "gpt-5.6-terra", + api: "openai-codex-responses", + provider: "openai-codex", + contextWindow: 272000, + }), + // The first-party API-key entry uses openai-responses and is untouched. + createSpec({ id: "gpt-5.6-sol", api: "openai-responses", provider: "openai", contextWindow: 1050000 }), + ]; + + applyGeneratedModelPolicies(models); + + expect(models[0]?.contextWindow).toBe(372000); + expect(models[1]?.contextWindow).toBe(372000); + expect(models[2]?.contextWindow).toBe(372000); + expect(models[3]?.contextWindow).toBe(1050000); + }); + it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => { const models: ModelSpec[] = [ createSpec({ diff --git a/packages/catalog/test/github-copilot-model-limits.test.ts b/packages/catalog/test/github-copilot-model-limits.test.ts index c21802a5d..36b5754c4 100644 --- a/packages/catalog/test/github-copilot-model-limits.test.ts +++ b/packages/catalog/test/github-copilot-model-limits.test.ts @@ -303,6 +303,66 @@ describe("github copilot model limits mapping", () => { // not the OpenAI global reference (1050k). expect(model?.contextWindow).toBe(272_000); }); + it("routes mai-code models to the openai-responses endpoint (#5612)", async () => { + // Copilot's /chat/completions rejects mai-* models with + // `unsupported_api_for_model` (400); they are served only via /responses. + const { models } = await discoverCopilotModels({ + data: [ + { + id: "mai-code-1-flash-picker", + name: "MAI-Code-1-Flash", + }, + ], + }); + + const model = models.find(candidate => candidate.id === "mai-code-1-flash-picker"); + expect(model).toBeDefined(); + expect(model?.api).toBe("openai-responses"); + }); + it("invalidates a cached MAI-Code completion route after the endpoint migration", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-copilot-mai-cache-")); + const cacheDbPath = path.join(tempDir, "models.db"); + const cacheProviderId = "github-copilot-mai-cache-test"; + try { + const oldManager = createModelManager({ + providerId: "github-copilot", + cacheProviderId, + cacheDbPath, + staticModels: [], + fetchDynamicModels: async () => [ + { + id: "mai-code-1-flash-picker", + name: "MAI-Code-1-Flash", + api: "openai-completions" as const, + provider: "github-copilot", + baseUrl: "https://api.githubcopilot.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 256_000, + maxTokens: 128_000, + }, + ], + }); + await oldManager.refresh("online"); + + const fetchMock = vi.fn(async () => { + throw new Error("a fresh cache must avoid discovery"); + }); + const manager = createModelManager({ + ...githubCopilotModelManagerOptions({ apiKey: "copilot-test-key", fetch: fetchMock }), + cacheProviderId, + cacheDbPath, + }); + const { models } = await manager.refresh("online-if-uncached"); + const model = models.find(candidate => candidate.id === "mai-code-1-flash-picker"); + + expect(fetchMock).not.toHaveBeenCalled(); + expect(model?.api).toBe("openai-responses"); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); }); /** diff --git a/packages/catalog/test/issue-1617-repro.test.ts b/packages/catalog/test/issue-1617-repro.test.ts index 101a2f93c..e80ae88da 100644 --- a/packages/catalog/test/issue-1617-repro.test.ts +++ b/packages/catalog/test/issue-1617-repro.test.ts @@ -38,23 +38,23 @@ describe("opencode-zen/-go resolver routes MiniMax M3 to openai-completions (iss const npmAnthropic: ModelsDevModel = { provider: { npm: "@ai-sdk/anthropic" }, tool_call: true }; describe("opencode-zen", () => { - test.each([ - ["minimax-m3"], - ["minimax-m3-free"], - ])("%s resolves to openai-completions on /v1/chat/completions", modelId => { - const resolved = zenDescriptor?.resolveApi?.(modelId, npmAnthropic); - expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_ZEN_BASE }); - }); + test.each([["minimax-m3"], ["minimax-m3-free"]])( + "%s resolves to openai-completions on /v1/chat/completions", + modelId => { + const resolved = zenDescriptor?.resolveApi?.(modelId, npmAnthropic); + expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_ZEN_BASE }); + }, + ); }); describe("opencode-go", () => { - test.each([ - ["minimax-m3"], - ["minimax-m3-free"], - ])("%s resolves to openai-completions on /v1/chat/completions", modelId => { - const resolved = goDescriptor?.resolveApi?.(modelId, npmAnthropic); - expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE }); - }); + test.each([["minimax-m3"], ["minimax-m3-free"]])( + "%s resolves to openai-completions on /v1/chat/completions", + modelId => { + const resolved = goDescriptor?.resolveApi?.(modelId, npmAnthropic); + expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE }); + }, + ); }); test("opencode-zen /v1/models refresh routes a freshly-discovered M3 to openai-completions", async () => { diff --git a/packages/catalog/test/issue-1849-repro.test.ts b/packages/catalog/test/issue-1849-repro.test.ts index 110e02cd6..8c4fff576 100644 --- a/packages/catalog/test/issue-1849-repro.test.ts +++ b/packages/catalog/test/issue-1849-repro.test.ts @@ -40,6 +40,12 @@ describe("Fireworks Kimi K2 maxTokens cap (#1849)", () => { "deepseek-v4-pro", "glm-5.1", "accounts/fireworks/models/minimax-m2.7", + // K2.7-Code is excluded from the K2.5/K2.6 cap (Fireworks serves its + // full context), so it must not match — public, fast, and wire ids. + "kimi-k2.7-code", + "kimi-k2.7-code-fast", + "kimi-k2.7-code-highspeed", + "accounts/fireworks/models/kimi-k2p7-code", ]; for (const id of negatives) { expect(isFireworksKimiK2ModelId(id)).toBe(false); @@ -71,4 +77,15 @@ describe("Fireworks Kimi K2 maxTokens cap (#1849)", () => { expect(model.maxTokens).toBe(FIREWORKS_KIMI_MAX_TOKENS); } }); + + it("leaves Kimi K2.7-Code uncapped on Fireworks", () => { + // K2.7-Code is not part of the K2.5/K2.6 cap; its output ceiling tracks + // Fireworks' reported max_completion_tokens rather than being pinned to + // the 32,768 family ceiling. + for (const id of ["kimi-k2.7-code", "kimi-k2.7-code-fast"]) { + const model = getBundledModel("fireworks", id); + expect(model).toBeDefined(); + expect(model.maxTokens).toBeGreaterThan(FIREWORKS_KIMI_MAX_TOKENS); + } + }); }); diff --git a/packages/catalog/test/issue-5572-repro.test.ts b/packages/catalog/test/issue-5572-repro.test.ts new file mode 100644 index 000000000..8533bea6b --- /dev/null +++ b/packages/catalog/test/issue-5572-repro.test.ts @@ -0,0 +1,151 @@ +import { describe, expect, it } from "bun:test"; +import { buildAnthropicClientOptions, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { Context, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; + +const CUSTOM_MODEL_SPEC: ModelSpec<"anthropic-messages"> = { + id: "claude-haiku-4.5", + name: "Claude Haiku 4.5", + api: "anthropic-messages", + provider: "internal-anthropic", + baseUrl: "https://llm.example.com/v1/messages", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, +}; + +const TOOLS: Tool[] = [ + { + name: "ping", + description: "ping", + parameters: { + type: "object", + properties: { msg: { type: "string" } }, + required: ["msg"], + } as TJsonSchema, + }, +]; + +const CONTEXT: Context = { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], + tools: TOOLS, +}; + +function aborted(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +describe("issue #5572 — custom Anthropic endpoints reject eager_input_streaming", () => { + it("omits eager_input_streaming from custom endpoint tool definitions", async () => { + const model = buildModel(CUSTOM_MODEL_SPEC); + const { promise, resolve } = Promise.withResolvers(); + streamAnthropic(model, CONTEXT, { + apiKey: "sk-ant-test", + signal: aborted(), + onPayload: payload => resolve(payload), + }); + + const payload = (await promise) as { tools?: Array> }; + expect(payload.tools).toHaveLength(1); + expect(payload.tools?.[0]).not.toHaveProperty("eager_input_streaming"); + }); + + it("omits the legacy fine-grained streaming beta from custom endpoint requests", () => { + const model = buildModel(CUSTOM_MODEL_SPEC); + const options = buildAnthropicClientOptions({ + model, + apiKey: "sk-ant-test", + extraBetas: [], + stream: true, + interleavedThinking: false, + hasTools: true, + }); + + expect(options.defaultHeaders["anthropic-beta"] ?? "").not.toContain("fine-grained-tool-streaming-2025-05-14"); + }); + + it("omits eager_input_streaming when a baseUrl-only override reroutes a canonical model", async () => { + // Mirrors `pi.registerProvider("anthropic", { baseUrl })`: the registry + // mutates `baseUrl` without rebuilding compat, so the resolved + // `supportsEagerToolInputStreaming` stays canonical-true. The authored + // spec never opted in, so the custom endpoint must not receive the flag. + const model = buildModel({ + ...CUSTOM_MODEL_SPEC, + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + }); + expect(model.compat.supportsEagerToolInputStreaming).toBe(true); + model.baseUrl = "https://proxy.example.com"; + + const { promise, resolve } = Promise.withResolvers(); + streamAnthropic(model, CONTEXT, { + apiKey: "sk-ant-test", + signal: aborted(), + onPayload: payload => resolve(payload), + }); + + const payload = (await promise) as { tools?: Array> }; + expect(payload.tools).toHaveLength(1); + expect(payload.tools?.[0]).not.toHaveProperty("eager_input_streaming"); + }); + + it("omits eager_input_streaming after a baseUrl override even when the spec baked a resolved official compat", async () => { + // Some bundled models (e.g. `claude-3-7-sonnet-20250219`) ship a + // fully-resolved compat block in models.json, so `compatConfig` carries + // `supportsEagerToolInputStreaming: true`. Gating on `compatConfig` alone + // would leak the field; the fix keys on the resolved `officialEndpoint` + // provenance instead. + const model = buildModel({ + ...CUSTOM_MODEL_SPEC, + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + compat: { supportsEagerToolInputStreaming: true }, + }); + expect(model.compat.officialEndpoint).toBe(true); + expect(model.compatConfig?.supportsEagerToolInputStreaming).toBe(true); + model.baseUrl = "https://proxy.example.com"; + + const { promise, resolve } = Promise.withResolvers(); + streamAnthropic(model, CONTEXT, { + apiKey: "sk-ant-test", + signal: aborted(), + onPayload: payload => resolve(payload), + }); + + const payload = (await promise) as { tools?: Array> }; + expect(payload.tools?.[0]).not.toHaveProperty("eager_input_streaming"); + }); + + it("honors explicit compat opt-in on a custom endpoint", async () => { + const model = buildModel({ + ...CUSTOM_MODEL_SPEC, + compat: { supportsEagerToolInputStreaming: true }, + }); + + const { promise, resolve } = Promise.withResolvers(); + streamAnthropic(model, CONTEXT, { + apiKey: "sk-ant-test", + signal: aborted(), + onPayload: payload => resolve(payload), + }); + + const payload = (await promise) as { tools?: Array> }; + expect(payload.tools?.[0]).toHaveProperty("eager_input_streaming", true); + }); + + it("keeps eager tool input streaming on the official Anthropic endpoint", () => { + const model = buildModel({ + ...CUSTOM_MODEL_SPEC, + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + }); + + expect(model.compat.supportsEagerToolInputStreaming).toBe(true); + }); +}); diff --git a/packages/catalog/test/issue-5598-repro.test.ts b/packages/catalog/test/issue-5598-repro.test.ts new file mode 100644 index 000000000..08390eb0a --- /dev/null +++ b/packages/catalog/test/issue-5598-repro.test.ts @@ -0,0 +1,56 @@ +import { describe, expect, test } from "bun:test"; +import { + MODELS_DEV_PROVIDER_DESCRIPTORS, + mapModelsDevToModels, +} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; + +// Z.AI GLM coding-plan token costs all showed as "Free" (issue #5598): the `zai` +// provider descriptor sourced the models.dev `zai-coding-plan` key, which reports +// all-$0 subscription rates. The `zai` (pay-as-you-go) key carries the real +// per-token rates for the identical GLM ids, matching how other subscription +// providers surface comparison pricing in `/models`. +describe("zai GLM pricing sources the PAYG models.dev key (issue #5598)", () => { + test("descriptor maps the `zai` models.dev key, not `zai-coding-plan`", () => { + const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "zai"); + expect(descriptor).toBeDefined(); + expect(descriptor?.modelsDevKey).toBe("zai"); + expect(descriptor?.api).toBe("anthropic-messages"); + expect(descriptor?.baseUrl).toBe("https://api.z.ai/api/anthropic"); + }); + + test("mapped zai models carry the PAYG per-token costs, not the coding-plan $0 rates", () => { + const payload = { + zai: { + models: { + "glm-5.2": { + name: "GLM-5.2", + reasoning: true, + tool_call: true, + modalities: { input: ["text"], output: ["text"] }, + cost: { input: 1.4, output: 4.4, cache_read: 0.26, cache_write: 0 }, + limit: { context: 1_000_000, output: 131_072 }, + }, + }, + }, + "zai-coding-plan": { + models: { + "glm-5.2": { + name: "GLM-5.2", + reasoning: true, + tool_call: true, + modalities: { input: ["text"], output: ["text"] }, + cost: { input: 0, output: 0, cache_read: 0, cache_write: 0 }, + limit: { context: 1_000_000, output: 131_072 }, + }, + }, + }, + }; + + const zai = mapModelsDevToModels(payload, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( + model => model.provider === "zai", + ); + const glm52 = zai.find(model => model.id === "glm-5.2"); + expect(glm52).toBeDefined(); + expect(glm52?.cost).toEqual({ input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 }); + }); +}); diff --git a/packages/catalog/test/issue-5756-repro.test.ts b/packages/catalog/test/issue-5756-repro.test.ts new file mode 100644 index 000000000..50452f157 --- /dev/null +++ b/packages/catalog/test/issue-5756-repro.test.ts @@ -0,0 +1,162 @@ +/** + * Issue #5756 — `moonshot/kimi-k3 is incorrectly shown as free` + * + * The native Moonshot `kimi-k3` entry is dynamically discovered but has no + * bundled/models.dev reference, so `mapWithBundledReference` produced the + * generic dynamic defaults: zero token cost, null limits, text-only input, + * and `reasoning: false`. `/models` then labeled the paid model "Free". + * + * The fix stamps Moonshot's official K3 pricing/limits, marks it reasoning + + * vision, and routes reasoning through OpenAI-style `reasoning_effort: "max"` + * (K3 does NOT accept the K2.x binary `thinking: { type }` block). + */ +import { describe, expect, it } from "bun:test"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import type { Context } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { moonshotModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; + +function moonshotModelsResponse(): Response { + const body = { + object: "list", + data: [ + { id: "kimi-k3", object: "model", owned_by: "moonshot" }, + { id: "kimi-k2.6", object: "model", owned_by: "moonshot" }, + ], + }; + return new Response(JSON.stringify(body), { + status: 200, + headers: { "content-type": "application/json" }, + }); +} + +async function discoverKimiK3(): Promise> { + const fetchMock = (async (_input: string | URL | Request): Promise => + moonshotModelsResponse()) as typeof fetch; + const models = await moonshotModelManagerOptions({ apiKey: "test-key", fetch: fetchMock }).fetchDynamicModels?.(); + const k3 = models?.find(m => m.id === "kimi-k3"); + if (!k3) throw new Error("kimi-k3 not discovered"); + return k3; +} + +function encodeSseChunks(chunks: ReadonlyArray>): string { + return `${chunks.map(c => `data: ${JSON.stringify(c)}\n\n`).join("")}data: [DONE]\n\n`; +} + +describe("issue #5756 — moonshot kimi-k3 pricing and wire format", () => { + it("discovery mapper stamps K3 pricing, limits, vision, and reasoning", async () => { + const k3 = await discoverKimiK3(); + expect(k3.cost).toEqual({ input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 }); + expect(k3.contextWindow).toBe(1_048_576); + expect(k3.maxTokens).toBe(131_072); + expect(k3.input).toEqual(["text", "image"]); + expect(k3.reasoning).toBe(true); + // The effort ladder itself tracks the generated catalog policies; the + // binding K3 contract — reasoning_effort=max on the wire, no K2-style + // thinking block — is asserted by the wire-body test below. + expect(k3.thinking?.mode).toBe("effort"); + expect(k3.thinking?.efforts?.length).toBeGreaterThan(0); + }); + + it("K3 native compat uses the OpenAI reasoning_effort dialect, not the K2 thinking block", async () => { + const model = buildModel(await discoverKimiK3()); + expect(model.compat.thinkingFormat).toBe("openai"); + expect(model.compat.reasoningDisableMode).toBe("lowest-effort"); + expect(model.compat.supportsReasoningEffort).toBe(true); + }); + + it("wire body carries reasoning_effort=max and omits the thinking block", async () => { + const model = buildModel(await discoverKimiK3()); + let body: Record = {}; + const fetchMock = (async (_input: string | URL | Request, init?: RequestInit): Promise => { + const raw = typeof init?.body === "string" ? init.body : ""; + body = raw ? (JSON.parse(raw) as Record) : {}; + return new Response( + encodeSseChunks([ + { choices: [{ index: 0, delta: { role: "assistant", content: "hi" }, finish_reason: null }] }, + { + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + usage: { prompt_tokens: 1, completion_tokens: 1 }, + }, + ]), + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); + }) as typeof fetch; + + const context: Context = { + messages: [{ role: "user", content: "hi", timestamp: Date.now() }], + }; + const stream = streamOpenAICompletions(model, context, { + apiKey: "test-key", + reasoning: "max", + fetch: fetchMock, + }); + for await (const _ of stream) { + // drain + } + + expect(body.reasoning_effort).toBe("max"); + expect("thinking" in body).toBe(false); + // Moonshot-native Kimi rate-limits on max_tokens, not + // max_completion_tokens; K3's default reaches its advertised 131K cap. + expect(body.max_tokens).toBe(131_072); + expect(body.max_completion_tokens).toBeUndefined(); + }); + + it("keeps reasoning_effort=max on forced-tool-choice turns (mandatory K3 reasoning)", async () => { + // K3 always reasons via `reasoning_effort: "max"`. The K2.x Kimi + // `disableReasoningOnForcedToolChoice` rule (Moonshot 400s on forced + // tool_choice + the binary `thinking` block, #827) must NOT strip K3's + // effort, or plan-mode `toolChoice` turns run without the required + // reasoning (#5758 review). + const model = buildModel(await discoverKimiK3()); + let body: Record = {}; + const fetchMock = (async (_input: string | URL | Request, init?: RequestInit): Promise => { + const raw = typeof init?.body === "string" ? init.body : ""; + body = raw ? (JSON.parse(raw) as Record) : {}; + return new Response( + encodeSseChunks([ + { + choices: [ + { + index: 0, + delta: { + role: "assistant", + tool_calls: [ + { index: 0, id: "c1", type: "function", function: { name: "plan", arguments: "{}" } }, + ], + }, + finish_reason: null, + }, + ], + }, + { + choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }], + usage: { prompt_tokens: 1, completion_tokens: 1 }, + }, + ]), + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); + }) as typeof fetch; + + const context: Context = { + messages: [{ role: "user", content: "hi", timestamp: Date.now() }], + tools: [{ name: "plan", description: "plan", parameters: { type: "object", properties: {} } }], + }; + const stream = streamOpenAICompletions(model, context, { + apiKey: "test-key", + reasoning: "max", + maxTokens: 131_072, + toolChoice: { type: "tool", name: "plan" }, + fetch: fetchMock, + }); + for await (const _ of stream) { + // drain + } + + expect(body.reasoning_effort).toBe("max"); + expect("thinking" in body).toBe(false); + expect(body.max_tokens).toBe(131_072); + }); +}); diff --git a/packages/catalog/test/issue-887-repro.test.ts b/packages/catalog/test/issue-887-repro.test.ts index 09f445c0e..e91d728a4 100644 --- a/packages/catalog/test/issue-887-repro.test.ts +++ b/packages/catalog/test/issue-887-repro.test.ts @@ -26,14 +26,13 @@ describe("opencode-go resolver routes 404-ing ids to openai-completions (issue # // would route them to /v1/messages on opencode.ai/zen/go which 404s. const npmAnthropic: ModelsDevModel = { provider: { npm: "@ai-sdk/anthropic" }, tool_call: true }; - test.each([ - ["minimax-m2.7"], - ["qwen3.5-plus"], - ["qwen3.6-plus"], - ])("%s resolves to openai-completions on /v1/chat/completions", modelId => { - const resolved = descriptor?.resolveApi?.(modelId, npmAnthropic); - expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE }); - }); + test.each([["minimax-m2.7"], ["qwen3.5-plus"], ["qwen3.6-plus"]])( + "%s resolves to openai-completions on /v1/chat/completions", + modelId => { + const resolved = descriptor?.resolveApi?.(modelId, npmAnthropic); + expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE }); + }, + ); test("minimax-m2.5 (control: works empirically) also resolves to openai-completions", () => { // models.dev currently lists minimax-m2.5 without an explicit provider.npm, diff --git a/packages/catalog/test/kimi-code-provider.test.ts b/packages/catalog/test/kimi-code-provider.test.ts new file mode 100644 index 000000000..50ea839e8 --- /dev/null +++ b/packages/catalog/test/kimi-code-provider.test.ts @@ -0,0 +1,78 @@ +import { describe, expect, it } from "bun:test"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { kimiCodeModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +const LIVE_K3 = { + id: "k3", + display_name: "K3", + context_length: 1_048_576, + supports_reasoning: true, + supports_thinking_type: "only", + think_efforts: { + support: true, + valid_efforts: ["low", "high", "max"], + default_effort: "max", + }, + protocol: null, +}; + +async function discover(models: readonly Record[]) { + const fetchImpl: FetchImpl = async () => Response.json({ data: models }); + const fetchDynamicModels = kimiCodeModelManagerOptions({ apiKey: "test-key", fetch: fetchImpl }).fetchDynamicModels; + if (!fetchDynamicModels) throw new Error("Kimi Code dynamic discovery is not configured"); + return (await fetchDynamicModels())?.map(buildModel) ?? []; +} + +describe("Kimi Code provider catalog", () => { + it("uses live K3 effort, mandatory-thinking, and native-protocol metadata", async () => { + const models = await discover([LIVE_K3]); + const model = models.find(candidate => candidate.id === "k3"); + + expect(model).toMatchObject({ + id: "k3", + name: "K3", + reasoning: true, + contextWindow: 1_048_576, + thinking: { + mode: "effort", + efforts: [Effort.Low, Effort.High, Effort.Max], + defaultLevel: Effort.Max, + requiresEffort: true, + }, + compat: { + thinkingFormat: "kimi", + kimiApiFormat: "openai", + }, + }); + }); + + it("uses server protocol while preserving legacy K2 discovery defaults", async () => { + const models = await discover([ + { ...LIVE_K3, id: "k3-anthropic", protocol: "anthropic" }, + { + id: "kimi-for-coding", + display_name: "K2.7 Code", + context_length: 262_144, + supports_reasoning: true, + }, + ]); + const anthropic = models.find(candidate => candidate.id === "k3-anthropic"); + const legacy = models.find(candidate => candidate.id === "kimi-for-coding"); + + expect(anthropic?.compat.kimiApiFormat).toBe("anthropic"); + expect(legacy?.compat).toMatchObject({ thinkingFormat: "zai" }); + expect(legacy?.compat.kimiApiFormat).toBeUndefined(); + expect(legacy?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]); + }); + + it("lets supports_thinking_type override the legacy reasoning flag", async () => { + const models = await discover([ + { ...LIVE_K3, id: "non-thinking", supports_thinking_type: "no", think_efforts: undefined }, + ]); + + expect(models[0]?.reasoning).toBe(false); + expect(models[0]?.thinking).toBeUndefined(); + }); +}); diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index 48ded780f..24183fd24 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -1,6 +1,7 @@ import { afterEach, describe, expect, test, vi } from "bun:test"; import { fetchLiteLLMRichModels, litellmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; +import * as logger from "@oh-my-pi/pi-utils/logger"; const ORIGINAL_LITELLM_BASE_URL = Bun.env.LITELLM_BASE_URL; const MODELS_DEV_URL = "https://models.dev/api.json"; @@ -125,7 +126,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v4:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, + `litellm:rich-v5:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -148,7 +149,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v4:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, + `litellm:rich-v5:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -238,6 +239,116 @@ describe("LiteLLM provider discovery", () => { }); }); + test("warns once when forbidden rich metadata forces /v1/models fallback", async () => { + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://forbidden:4000/v1/models") { + return Response.json({ data: [{ id: "hosted_vllm/private-model" }] }); + } + return new Response("Forbidden", { status: 403 }); + }) as FetchImpl; + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + const options = litellmModelManagerOptions({ + apiKey: "sk-restricted", + baseUrl: "http://forbidden:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + await options.fetchDynamicModels?.(); + + expect(models?.[0]).toMatchObject({ + id: "hosted_vllm/private-model", + contextWindow: null, + maxTokens: null, + }); + expect(warnSpy).toHaveBeenCalledTimes(1); + expect(warnSpy).toHaveBeenCalledWith( + "LiteLLM rich model metadata unavailable; falling back to /v1/models", + expect.objectContaining({ + endpoint: "http://forbidden:4000/model_group/info", + status: 403, + reason: "http-status", + }), + ); + }); + + test("treats missing rich metadata endpoints as absent without warning", async () => { + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + const models = await fetchLiteLLMRichModels({ + api: "openai-completions", + provider: "litellm", + apiKey: "sk-restricted", + baseUrl: "http://missing:4000/v1", + fetch: async () => new Response("Not Found", { status: 404 }), + }); + + expect(models).toBeNull(); + expect(warnSpy).not.toHaveBeenCalled(); + }); + + test("stays silent on retryable 401 rich failures so the caller's auth retry owns them", async () => { + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + const models = await fetchLiteLLMRichModels({ + api: "openai-completions", + provider: "litellm", + apiKey: "sk-stale", + baseUrl: "http://unauthorized:4000/v1", + fetch: async () => new Response("Unauthorized", { status: 401 }), + }); + + expect(models).toBeNull(); + expect(warnSpy).not.toHaveBeenCalled(); + }); + + test("maps LiteLLM per-token cost onto cost.input/output for models missing from models.dev", async () => { + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ + data: [ + { + model_group: "openrouter/acme/big", + model_name: "Acme Big", + max_input_tokens: 262_144, + max_output_tokens: 16_384, + input_cost_per_token: 0.000_005, + output_cost_per_token: 0.000_03, + cache_read_input_token_cost: 0.000_000_5, + supports_vision: true, + }, + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when rich metadata succeeds"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(models).toHaveLength(1); + expect(models?.[0]).toMatchObject({ + id: "openrouter/acme/big", + contextWindow: 262_144, + cost: { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 0 }, + }); + }); + test("enriches LiteLLM rich models missing from models.dev with bundled reasoning metadata", async () => { const fetchMock = vi.fn(async (input: string | URL | Request) => { const url = inputUrl(input); @@ -317,61 +428,60 @@ describe("LiteLLM provider discovery", () => { expect(models?.find(model => model.id === "params-tools")?.supportsTools).toBe(true); }); - test.each([ - ["all-team-models"], - ["all-proxy-models"], - ["no-default-models"], - ])("falls back from %s placeholder to v2 model info", async sentinelModelId => { - const calls: string[] = []; - const fetchMock = vi.fn(async (input: string | URL | Request) => { - const url = inputUrl(input); - calls.push(url); - if (url === MODELS_DEV_URL) { - return Response.json({}); - } - if (url === "http://primary:4000/model_group/info") { - return Response.json({ data: [makeLiteLLMSentinelPlaceholder(sentinelModelId)] }); - } - if (url === "http://primary:4000/v2/model/info") { - return Response.json({ - data: [ - { - model_name: "example-real-model", - model_info: { - max_input_tokens: 200_000, - max_output_tokens: 12_000, - supports_vision: false, - supports_reasoning: true, + test.each([["all-team-models"], ["all-proxy-models"], ["no-default-models"]])( + "falls back from %s placeholder to v2 model info", + async sentinelModelId => { + const calls: string[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + calls.push(url); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ data: [makeLiteLLMSentinelPlaceholder(sentinelModelId)] }); + } + if (url === "http://primary:4000/v2/model/info") { + return Response.json({ + data: [ + { + model_name: "example-real-model", + model_info: { + max_input_tokens: 200_000, + max_output_tokens: 12_000, + supports_vision: false, + supports_reasoning: true, + }, }, - }, - ], - }); - } - if (url === "http://primary:4000/v1/models") { - throw new Error("/v1/models should not be called when v2 metadata succeeds"); - } - throw new Error(`Unexpected URL: ${url}`); - }) as FetchImpl; - const options = litellmModelManagerOptions({ - apiKey: "sk-rich", - baseUrl: "http://primary:4000/v1", - fetch: fetchMock, - }); + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when v2 metadata succeeds"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); - const models = await options.fetchDynamicModels?.(); + const models = await options.fetchDynamicModels?.(); - expect(calls).toContain("http://primary:4000/model_group/info"); - expect(calls).toContain("http://primary:4000/v2/model/info"); - expect(calls).not.toContain("http://primary:4000/v1/models"); - expect(models?.map(model => model.id)).toEqual(["example-real-model"]); - expect(models?.[0]).toMatchObject({ - id: "example-real-model", - contextWindow: 200_000, - maxTokens: 12_000, - input: ["text"], - reasoning: true, - }); - }); + expect(calls).toContain("http://primary:4000/model_group/info"); + expect(calls).toContain("http://primary:4000/v2/model/info"); + expect(calls).not.toContain("http://primary:4000/v1/models"); + expect(models?.map(model => model.id)).toEqual(["example-real-model"]); + expect(models?.[0]).toMatchObject({ + id: "example-real-model", + contextWindow: 200_000, + maxTokens: 12_000, + input: ["text"], + reasoning: true, + }); + }, + ); test("filters all-team-models placeholder from mixed model_group info", async () => { const calls: string[] = []; diff --git a/packages/catalog/test/lm-studio-provider.test.ts b/packages/catalog/test/lm-studio-provider.test.ts index 6b8aee0fd..ab76be313 100644 --- a/packages/catalog/test/lm-studio-provider.test.ts +++ b/packages/catalog/test/lm-studio-provider.test.ts @@ -48,6 +48,54 @@ describe("lm studio local provider discovery", () => { expect(text?.input).toEqual(["text"]); }); + test("prefers the loaded context window over the architectural maximum", async () => { + const fetchMock: FetchImpl = vi.fn(async input => { + const url = String(input); + if (url === "http://127.0.0.1:1234/api/v0/models") { + return new Response( + JSON.stringify({ + data: [ + { + id: "loaded-small", + type: "llm", + state: "loaded", + max_context_length: 262144, + loaded_context_length: 81920, + }, + { + id: "unloaded", + type: "llm", + state: "not-loaded", + max_context_length: 262144, + loaded_context_length: null, + }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + if (url === "http://127.0.0.1:1234/v1/models") { + return new Response( + JSON.stringify({ + data: [ + { id: "loaded-small", object: "model" }, + { id: "unloaded", object: "model" }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }); + + const models = await lmStudioModelManagerOptions({ fetch: fetchMock }).fetchDynamicModels?.(); + const loaded = models?.find(model => model.id === "loaded-small"); + const unloaded = models?.find(model => model.id === "unloaded"); + + expect(loaded?.contextWindow).toBe(81920); + expect(unloaded?.contextWindow).toBe(262144); + }); + test("falls back to the OpenAI-compatible catalog when native metadata hangs", async () => { let nativeAborted = false; let openAiCatalogStartedBeforeAbort = false; diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 5c12451f6..650d87bc8 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -584,6 +584,31 @@ describe("model thinking derivation", () => { expect(fable.compat.supportsSamplingParams).toBe(false); }); + it("bakes sampling-param rejection into OpenAI reasoning compat (#5606)", () => { + // GitHub Copilot Responses gpt-5.6 — the reported failing model. + const luna = createModel({ + id: "gpt-5.6-luna", + api: "openai-responses", + provider: "github-copilot", + baseUrl: "https://api.githubcopilot.com", + }); + const gpt5 = createModel({ id: "gpt-5", api: "openai-responses", provider: "openai" }); + const gpt5Mini = createModel({ id: "gpt-5-mini", api: "openai-completions", provider: "openai" }); + const gpt5Chat = createModel({ id: "gpt-5-chat-latest", api: "openai-responses", provider: "openai" }); + const oThree = createModel({ id: "o3-mini", api: "openai-responses", provider: "openai" }); + // Non-restricted OpenAI + non-OpenAI models keep sampling support. + const gpt4o = createModel({ id: "gpt-4o", api: "openai-responses", provider: "openai", reasoning: false }); + const kimi = createModel({ id: "kimi-k2.6", api: "openai-completions", provider: "moonshot" }); + + expect(luna.compat.supportsSamplingParams).toBe(false); + expect(gpt5.compat.supportsSamplingParams).toBe(false); + expect(gpt5Mini.compat.supportsSamplingParams).toBe(false); + expect(gpt5Chat.compat.supportsSamplingParams).toBe(false); + expect(oThree.compat.supportsSamplingParams).toBe(false); + expect(gpt4o.compat.supportsSamplingParams).toBe(true); + expect(kimi.compat.supportsSamplingParams).toBe(true); + }); + it("encodes effort-dial-less reasoners as thinking: undefined", () => { const model = createModel({ id: "grok-build", diff --git a/packages/catalog/test/umans-provider.test.ts b/packages/catalog/test/umans-provider.test.ts index a641e6ec9..197ac6431 100644 --- a/packages/catalog/test/umans-provider.test.ts +++ b/packages/catalog/test/umans-provider.test.ts @@ -12,24 +12,7 @@ import { import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; import modelsJson from "../src/models.json"; -interface BundledModel { - api: string; - provider: string; - baseUrl: string; - reasoning: boolean; - input: string[]; - contextWindow: number | null; - maxTokens: number | null; - thinking?: { - defaultLevel?: string; - requiresEffort?: boolean; - efforts?: string[]; - effortMap?: Record; - }; - compat?: { - escapeBuiltinToolNames?: boolean; - }; -} +const bundledModels = modelsJson; describe("umans provider catalog", () => { it("discovers Anthropic-route models from the public models info endpoint", async () => { @@ -98,6 +81,7 @@ describe("umans provider catalog", () => { baseUrl: "https://api.code.umans.ai", reasoning: true, input: ["text", "image"], + cost: { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 }, contextWindow: 262_144, maxTokens: 32_768, thinking: { defaultLevel: "medium" }, @@ -179,8 +163,7 @@ describe("umans provider catalog", () => { }); it("bundles Umans GLM via-handoff models as text-only", () => { - const providers = modelsJson as Record>; - const model = providers.umans?.["umans-glm-5.2"]; + const model = bundledModels.umans?.["umans-glm-5.2"]; expect(model, "umans-glm-5.2 should be bundled").toBeDefined(); expect(model.input, "umans-glm-5.2 input should be text-only").toEqual(["text"]); }); @@ -245,9 +228,21 @@ describe("umans provider catalog", () => { } }); - it("maps the models.dev Umans provider to the Anthropic endpoint", () => { + it("maps the models.dev Umans PAYG pricing to the Anthropic endpoint", () => { const models = mapModelsDevToModels( { + "umans-ai": { + models: { + "umans-coder": { + name: "Umans Coder", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "image"] }, + limit: { context: 262_144, output: 262_144 }, + cost: { input: 0.95, output: 4, cache_read: 0.19 }, + }, + }, + }, "umans-ai-coding-plan": { models: { "umans-coder": { @@ -272,14 +267,14 @@ describe("umans provider catalog", () => { baseUrl: "https://api.code.umans.ai", reasoning: true, input: ["text", "image"], + cost: { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 }, contextWindow: 262_144, maxTokens: 262_144, }); }); it("bundles the default Umans coding model", () => { - const providers = modelsJson as Record>; - const model = providers.umans?.["umans-coder"]; + const model = bundledModels.umans?.["umans-coder"]; expect(model).toBeDefined(); expect(model).toMatchObject({ @@ -294,9 +289,28 @@ describe("umans provider catalog", () => { }); }); + it("bundles published Umans PAYG pricing", () => { + const models = bundledModels.umans; + + expect(models?.["umans-coder"].cost).toEqual({ input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 }); + expect(models?.["umans-kimi-k2.7"].cost).toEqual({ + input: 0.95, + output: 4, + cacheRead: 0.19, + cacheWrite: 0, + }); + expect(models?.["umans-glm-5.2"].cost).toEqual({ input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 }); + expect(models?.["umans-flash"].cost).toEqual({ input: 0.15, output: 1, cacheRead: 0.05, cacheWrite: 0 }); + expect(models?.["umans-qwen3.6-35b-a3b"].cost).toEqual({ + input: 0.15, + output: 1, + cacheRead: 0.05, + cacheWrite: 0, + }); + }); + it("bundles Umans mandatory reasoning metadata", () => { - const providers = modelsJson as Record>; - const model = providers.umans?.["umans-kimi-k2.7"]; + const model = bundledModels.umans?.["umans-kimi-k2.7"]; expect(model).toBeDefined(); expect(model.maxTokens).toBe(32_768); @@ -307,14 +321,13 @@ describe("umans provider catalog", () => { }); it("bundles Umans GLM 5.2 with the wire-exact high/max ladder", () => { - const providers = modelsJson as Record>; - const model = providers.umans?.["umans-glm-5.2"]; + const model = bundledModels.umans?.["umans-glm-5.2"]; expect(model).toBeDefined(); expect(model.thinking).toMatchObject({ mode: "anthropic-budget-effort", efforts: ["high", "max"], }); - expect(model.thinking?.effortMap).toBeUndefined(); + expect("effortMap" in model.thinking).toBe(false); }); }); diff --git a/packages/catalog/test/variant-collapse.test.ts b/packages/catalog/test/variant-collapse.test.ts index cbfc93781..6b1724808 100644 --- a/packages/catalog/test/variant-collapse.test.ts +++ b/packages/catalog/test/variant-collapse.test.ts @@ -109,6 +109,41 @@ describe("collapseEffortVariants", () => { }); }); + it("collapses Gemini 3.6 Flash tiers into one routed logical spec", () => { + const out = collapseEffortVariants( + [ + memberSpec("gemini-3.6-flash-high"), + memberSpec("gemini-3.6-flash-low"), + memberSpec("gemini-3.6-flash-medium"), + memberSpec("gemini-3.6-flash-tiered"), + ], + ANTIGRAVITY_VARIANT_COLLAPSE_TABLE, + ); + + expect(out).toHaveLength(1); + const flash = out[0]; + expect(flash?.id).toBe("gemini-3.6-flash"); + expect(flash?.name).toBe("Gemini 3.6 Flash"); + expect(flash?.requestModelId).toBe("gemini-3.6-flash-low"); + expect(flash?.thinking).toEqual({ + mode: "google-level", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High], + requiresEffort: true, + effortRouting: { + minimal: "gemini-3.6-flash-low", + low: "gemini-3.6-flash-low", + medium: "gemini-3.6-flash-medium", + high: "gemini-3.6-flash-high", + }, + }); + + const model = buildModel(flash as ModelSpec<"google-gemini-cli">); + expect(resolveWireModelId(model, Effort.Minimal)).toBe("gemini-3.6-flash-low"); + expect(resolveWireModelId(model, Effort.Low)).toBe("gemini-3.6-flash-low"); + expect(resolveWireModelId(model, Effort.Medium)).toBe("gemini-3.6-flash-medium"); + expect(resolveWireModelId(model, Effort.High)).toBe("gemini-3.6-flash-high"); + }); + it("drops routes whose target member is absent", () => { const out = collapseEffortVariants( [memberSpec("gemini-3.5-flash-extra-low")], @@ -791,3 +826,75 @@ describe("antigravity discovery collapsing", () => { expect(models?.[0]?.baseUrl).toBe(ANTIGRAVITY_PRIMARY_ENDPOINT); }); }); + +describe("Devin GLM-5.2 collapse", () => { + function devinMemberSpec(id: string, overrides: Partial> = {}): ModelSpec<"devin-agent"> { + return { + id, + name: id, + api: "devin-agent", + provider: "devin", + baseUrl: "https://server.codeium.com", + reasoning: true, + input: ["text"], + supportsTools: true, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 64_000, + ...overrides, + }; + } + + it("collapses the three 200K GLM-5.2 variants into one logical entry routing all efforts to the free glm-5-2 wire UID", () => { + const out = collapseEffortVariants( + [ + devinMemberSpec("glm-5-2"), + devinMemberSpec("glm-5-2-max"), + devinMemberSpec("glm-5-2-none", { reasoning: false }), + ], + DEVIN_VARIANT_COLLAPSE_TABLE, + ); + + expect(out).toHaveLength(1); + const spec = out[0]; + expect(spec?.id).toBe("glm-5-2"); + expect(spec?.thinking?.effortRouting).toEqual({ + high: "glm-5-2", + xhigh: "glm-5-2", + }); + }); + + it("routes every effort to glm-5-2 (never to the quota-gated glm-5-2-max or glm-5-2-none)", () => { + const out = collapseEffortVariants( + [devinMemberSpec("glm-5-2"), devinMemberSpec("glm-5-2-max")], + DEVIN_VARIANT_COLLAPSE_TABLE, + ); + + const spec = out[0]; + const routing = spec?.thinking?.effortRouting ?? {}; + for (const wire of Object.values(routing)) { + expect(wire).toBe("glm-5-2"); + } + }); + + it("collapses the three 1M GLM-5.2 variants into one paid entry with proper effort routing", () => { + const out = collapseEffortVariants( + [ + devinMemberSpec("glm-5-2-1m", { contextWindow: 1_000_000 }), + devinMemberSpec("glm-5-2-max-1m", { contextWindow: 1_000_000 }), + devinMemberSpec("glm-5-2-none-1m", { contextWindow: 1_000_000, reasoning: false }), + ], + DEVIN_VARIANT_COLLAPSE_TABLE, + ); + + expect(out).toHaveLength(1); + const spec = out[0]; + expect(spec?.id).toBe("glm-5-2-1m"); + expect(spec?.contextWindow).toBe(1_000_000); + expect(spec?.thinking?.effortRouting).toEqual({ + off: "glm-5-2-none-1m", + high: "glm-5-2-1m", + xhigh: "glm-5-2-max-1m", + }); + }); +}); diff --git a/packages/catalog/test/zhipu-compat.test.ts b/packages/catalog/test/zhipu-compat.test.ts index 82ff798be..b69b9afb7 100644 --- a/packages/catalog/test/zhipu-compat.test.ts +++ b/packages/catalog/test/zhipu-compat.test.ts @@ -120,6 +120,39 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { }); }); +describe("openai-completions compat — GLM coding-plan stream idle timeout", () => { + function glm52(provider: string, baseUrl: string): ModelSpec<"openai-completions"> { + return { ...baseModel, id: "glm-5.2", name: "GLM-5.2", provider, baseUrl }; + } + + // GLM coding-plan SKUs idle for minutes mid-reasoning; the 600s watchdog + // floor must apply on every gateway that fronts them, not just the native + // Z.AI/Zhipu hosts (issue #4758: GLM-5.2 via opencode-go stalled with + // "OpenAI completions stream stalled while waiting for the next event"). + it("widens the idle timeout to 600s for GLM-5.x on Z.AI, Zhipu, and OpenCode gateways", () => { + expect(buildOpenAICompat(glm52("zai", "https://api.z.ai/api/coding/paas/v4")).streamIdleTimeoutMs).toBe(600_000); + expect( + buildOpenAICompat(glm52("zhipu-coding-plan", "https://open.bigmodel.cn/api/coding/paas/v4")) + .streamIdleTimeoutMs, + ).toBe(600_000); + expect(buildOpenAICompat(glm52("opencode-go", "https://opencode.ai/zen/go/v1")).streamIdleTimeoutMs).toBe( + 600_000, + ); + expect(buildOpenAICompat(glm52("opencode-zen", "https://opencode.ai/zen/v1")).streamIdleTimeoutMs).toBe(600_000); + }); + + it("does not widen non-GLM models on the OpenCode gateway via the GLM floor", () => { + const kimi = buildOpenAICompat({ + ...baseModel, + id: "kimi-k2.5", + name: "Kimi K2.5", + provider: "opencode-go", + baseUrl: "https://opencode.ai/zen/go/v1", + }); + expect(kimi.streamIdleTimeoutMs).toBeUndefined(); + }); +}); + describe("zhipu-coding-plan model discovery", () => { it("uses the dedicated Coding Plan endpoint by default", async () => { let requestedUrl = ""; diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1c1ee839f..9d7418f20 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,6 +5,489 @@ ### Fixed - Fixed Ctrl-clicking a wrapped OAuth authorization URL opening only the clicked row's truncated fragment by preserving the complete hyperlink target on every rendered row. +### Added + +- Added per-call `model` selection to the `task` tool, including per-item batch selectors, fallback chains, and explicit reasoning suffixes. +- Added Firecrawl keyless mode: explicitly selecting `firecrawl` as the web-search provider now works without `FIRECRAWL_API_KEY` by calling the Firecrawl REST API without an `Authorization` header; the automatic provider chain remains credential-gated (#4332). +- Added `mcp.renderMarkdownResults` (enabled by default): non-JSON MCP text results render as Markdown in the terminal transcript; set it to `false` to keep raw text. + +### Changed + +- Adjusted retry fallback handling to recognize discovery-only and runtime extension providers, preventing spurious unknown-provider warnings. + +### Fixed + +- Fixed the setup wizard hiding the selected row on short terminals (e.g. 24x80): the provider sign-in, theme, and web-search lists now fit their windows to the visible height, and decorative chrome (sign-in hint, theme mock preview) yields to the list when space is tight. +- Fixed restored sessions replaying terminal aborted or errored assistant turns, which could repeatedly fail continuation from an assistant role; `/retry` now consults the persisted transcript so the failed turn remains retryable without re-entering provider context. +- Fixed `get_available_models` and `set_model` RPCs racing background model discovery on cold start by awaiting the in-flight refresh before reading the registry. RPC/ACP clients that query the catalog or select a model immediately after session ready previously saw only statically-bundled models until discovery completed seconds later. +- Fixed deferred `--model /` CLI resolution failing on cold start with "Model not found" when the selector pointed at a discovery-backed provider (proxy / ollama / lm-studio / llama.cpp / litellm). The deferred retry now runs a cache-aware discovery pass before resolving, mirroring the default-role fallback's cold-cache race fix (issues #6114, #6162). +- Fixed MCP tool calls that return a `WWW-Authenticate` challenge by preserving the structured metadata, completing the configured OAuth flow, and retrying the call once on the refreshed connection. +- Fixed the Hindsight API token setting being absent from the Memory tab, so authenticated servers can be configured entirely in the TUI. +- Fixed aborted-task follow-up hints pointing at `history://` transcripts that cannot resolve: the hint now reports the transcript as unavailable when the agent ref retains no session file, while still-resumable agents keep their `hub` resume hint. +- Fixed compiled binaries failing to load legacy Pi extensions with minified imports, `pi-ai/compat`, or transitive runtime dependencies. The compatibility loader now follows compact static imports, resolves transitive on-disk ESM imports and CommonJS requires with package conditions, and restores the legacy `copyToClipboard` and `decodeKittyPrintable` root exports used by `pi-vimmode` and `pi-web-access`. +- Fixed a budget-aborted keep-alive subagent becoming an unkillable registration with no `hub`-level stop. A subagent force-stopped for exceeding its soft request budget is kept resumable (status `idle`, adopted by the lifecycle) so its context can be salvaged, but its async job row settles and is reaped after ~5 min — after which `hub cancel ` could only report `Background job not found` because it consulted the job manager alone. `hub cancel` now falls through to the agent registration: for an id the caller spawned that has no live job, it aborts any in-flight turn, disposes the session, and drops the registration (the interactive Agent Hub `x` and collab `kill` already did this; the model-facing `hub` did not). Cross-agent kills stay impossible and Main/advisor refs are never targeted ([#6315](https://github.com/can1357/oh-my-pi/issues/6315)). +- Fixed Agent Hub fallback rows hiding routing provenance and the resolved provider/model ([#6316](https://github.com/can1357/oh-my-pi/issues/6316)). +- Reduced format-on-write latency by avoiding cold language-server startup when diagnostics are disabled. +- Rewrote the `/guided-goal` interviewer rubric around loop-engineering: deterministic success criteria, verification commands, attempt caps, scope boundaries, and stop conditions. Ready objectives must use the five-section structured markdown form. +- Added `task.isolation.apply` (default `true`) to choose whether successful isolated `task` runs automatically apply their changes to the parent checkout or retain patch/branch artifacts for later integration. +- Added opt-in RPC protocol v2 negotiation with bounded, lossless chunking for stdout objects up to 64 MiB, plus stable cursor-based message pages for histories that should not travel as one response. Legacy JSONL clients remain on protocol v1, while the bundled TypeScript and Python RPC clients negotiate, reassemble, and drain message pages automatically. +- Fixed protocol v2 chunked framing materializing the whole base64 transport in memory: near-limit logical frames (~63 MiB) peaked around 686 MB RSS and over-ceiling frames allocated the full payload buffer before rejection. Chunk lines are now produced lazily from a single serialization, the 64 MiB ceiling is checked before any full-payload allocation, and RPC stdout writes honor backpressure line by line. +- Fixed the bundled TypeScript and Python RPC clients throwing when a `get_messages_page` cursor went stale mid-walk (e.g. a background bash appending a message between pages): the high-level `getMessages()` drains now discard partial pages and fall back to the legacy snapshot on both `session_busy` and `stale_cursor`, driven by a new machine-readable `code` field on RPC error responses. Direct page calls remain strict. + +## [17.0.8] - 2026-07-22 + +### Added + +- Added a `/tree` re-answer option for past `ask` tool results, allowing users to re-open the picker with original questions and branch the new answer as a sibling while keeping the original branch reachable. +- Added configurable Hindsight client request deadlines via `hindsight.requestTimeoutMs`, `reflectTimeoutMs`, `recallTimeoutMs`, and `retainTimeoutMs` settings (and matching `HINDSIGHT_*_TIMEOUT_MS` environment variables). +- Added `omp-linux-musl-x64` and `omp-linux-musl-arm64` release binaries for Alpine and other musl-based Linux distributions, with automatic musl selection in the installer and self-updater. + +### Changed + +- Optimized edit-tool previews, diff components, and intra-line word highlighting to compute line and word diffs natively, reducing synchronous diff times by 2-10x on large inputs. +- Updated diff generation and rendering components to rely exclusively on native UTF-16 diff bindings, removing `isWellFormed()` guards and JS fallback code paths. + +### Removed + +- Removed npm `diff` dependency. +- Fixed an issue where `Ctrl+V` clipboard paste was ignored while API-key and other modal prompts had focus. +- Fixed `scripts/install.sh` incorrectly installing an x86_64 build on Apple Silicon when running under Rosetta. +- Fixed the model picker hiding Codex models available through secondary configured ChatGPT/Codex OAuth accounts by unioning catalogs across all stored accounts. +- Fixed GitHub Copilot 1M-context models disappearing from the model picker on restart with a "Could not restore model" warning. +- Fixed `--model ` startup selection skipping configured fallback chains when the primary model is unavailable. +- Fixed global model role updates clobbering concurrent or external edits to `config.yml` by merging only the changed role instead of persisting a stale in-memory snapshot. +- Fixed terminal provider errors on continuation turns after failed tool results silently ending runs without persisting the error diagnostics. +- Fixed repeated OpenRouter Gemini stream closures consuming the full retry budget by limiting recovery attempts before surfacing the error. +- Fixed Agent Hub performance freezes when opening large read-only Advisor transcripts by collapsing synthetic inputs into compact summary rows and rendering Markdown lazily on expansion. +- Fixed `/agents` incorrectly showing prewalk as disabled for the bundled `task` agent when enabled by its runtime default. +- Fixed `hub start` waiting for the full timeout when a launched process exited or became ready quickly. +- Fixed `tools.maxTimeout` failing to clamp default tool timeouts when no explicit timeout was provided by the agent. +- Fixed MCP argument-shaping parity between direct and subagent tool calls, ensuring strict servers do not reject proxied calls with unrecognized keys. +- Fixed a crash in extensions like `pi-mcp-adapter` caused by the TypeBox compatibility shim omitting `Type.Unsafe`. +- Fixed `omp models` hanging after output by properly clearing managed extension timers and shutting down sessions before returning. +- Fixed HTML session exports causing browser call stack overflows when rendering deeply nested conversation trees. +- Fixed task agents ending prematurely on connection errors instead of entering the auto-retry path. +- Fixed a startup race condition where the default model role was incorrectly overridden by an unrelated provider's default on a cold cache. +- Fixed unqualified `--model` startup selection preferring unauthenticated provider catalog entries over configured providers. +- Fixed provider stream failures being invisible in the main log by logging a warning with error details when a turn ends in a provider error. +- Fixed parallel `todo done` calls losing completions due to asynchronous session events overwriting newer tool states. +- Fixed `omp` crashing when `git` is not installed or missing from the system `PATH`. +- Fixed `/changelog` commands reporting no entries in standalone binaries by embedding the release history as a fallback. +- Fixed isolated branch merge-backs rejecting committed agent edits when the parent branch had unrelated uncommitted changes in the same file. +- Fixed the 30-second Hindsight client timeout aborting healthy `reflect` operations by applying dedicated, longer deadlines. +- Fixed Mnemopi consolidation redundantly re-storing cumulative session transcripts after incremental auto-retain. +- Fixed turn-ending Codex rate-limit errors being hidden behind the Plan Review overlay. +- Fixed prewalked subagents continuing to display their starting model after switching to the target model. +- Fixed the Escape key aborting an ongoing agent turn instead of stopping text-to-speech playback. +- Fixed project system prompts shortening working directories to `~`, which could cause models to generate incorrect absolute paths for tool calls. +- Fixed the TUI `/usage` matrix misaligning multi-account columns across quota windows. +- Fixed near-miss `xd://` write targets silently creating filesystem paths instead of throwing a corrective URI error. +- Fixed the `task` tool rejecting valid batch calls with misleading validation errors when batching is disabled. +- Fixed JS/TS `debug` launches timing out on WSL2 with mirrored networking by waiting for the adapter's listening banner and handling transport closures immediately. +- Fixed post-compaction transcript rebuilds blocking the main thread by reusing settled message components and layout caches. +- Fixed the fullscreen Plan Review overlay remaining interactive and appearing frozen during slow asynchronous operations by locking input and showing a submitting indicator. +- Fixed dynamic model discovery refreshes dropping provider-level compatibility overrides from `models.yml`. +- Fixed a startup crash that locked users out of the app when `prewalk.enabled` was set but the prewalk hand-off target had no configured API key. +- Fixed in-progress aborts awaiting `session_stop` extension handlers whose results would be discarded. +- Fixed `/retry` reporting "Nothing to retry" after a stream stalled or aborted mid-tool-call. +- Fixed locally consumed extension commands triggering automatic title generation and exposing their command text to the title model. + +## [17.0.7] - 2026-07-21 + +### Fixed + +- Fixed Portkey/gateway custom models whose ids start with `@` (e.g. `@modal/GLM-5-2-FP8`) being rewritten to unrelated bundled wire ids (e.g. `glm-5-2`), which caused `400` responses requiring `x-portkey-config` or `x-portkey-provider`. + +## [17.0.6] - 2026-07-20 + +- Fixed failed plan-mode exits leaving the session on the restored execution model while plan mode remained active and silently changing ambient `xd://` tool presentation; rollback now restores the plan model, thinking level, and exact top-level-versus-mounted tool partition so exit can be retried safely ([#6013](https://github.com/can1357/oh-my-pi/pull/6013)). + +### Added + +- Added native Warp CLI-agent events for rich session status, tool approvals, and completion notifications ([#5592](https://github.com/can1357/oh-my-pi/pull/5592) by [@metaphorics](https://github.com/metaphorics)). +- Added Codex (ChatGPT subscription) support to `generate_image`. The tool now resolves a connected `openai-codex` OAuth credential and drives OpenAI's hosted `image_generation` tool through the ChatGPT backend (`chatgpt.com/backend-api/codex/responses`, `chatgpt-account-id` header) **independent of the active chat model** — so image generation works on a ChatGPT/Codex subscription with no metered `OPENAI_API_KEY`, even when the active model is Claude/Gemini/etc. A new `providers.image: "openai-codex"` option forces it; `auto` now auto-detects a connected subscription (priority: active GPT image tool > Codex subscription > Antigravity > xAI > OpenRouter > Gemini), and the `openai` preference falls back to it when no `OPENAI_API_KEY`/active GPT model is present. +- Added an optional `provider` parameter to `generate_image` (`auto` | `openai` | `openai-codex` | `antigravity` | `xai` | `gemini` | `openrouter`) that overrides the `providers.image` setting **for a single request** — so "generate this using gemini / codex / xai" routes per-call without changing the global setting. Absent → the `providers.image` setting applies, unchanged; the named provider uses the same resolution semantics (falls back to auto-detect if it has no credentials). File: `tools/image-gen.ts` (`imageProviderSchema`, `findImageApiKey` `preference` arg). +- Added OpenTelemetry log and metric export alongside the existing trace export. When `OTEL_EXPORTER_OTLP_LOGS_ENDPOINT` (or the shared `OTEL_EXPORTER_OTLP_ENDPOINT`) is set, `omp` registers a `LoggerProvider` and forwards every centralized-logger event as an OTLP log record (severity + attributes + active span context for log↔trace correlation, min level via `OTEL_LOG_LEVEL`, plus a structured `agent run completed` summary event). When `OTEL_EXPORTER_OTLP_METRICS_ENDPOINT` (or the shared endpoint) is set, it registers a `MeterProvider` with a `PeriodicExportingMetricReader` and records GenAI-semconv `gen_ai.client.token.usage` plus `pi.omp.agent.*` counters/histograms (runs, steps, chat/tool calls by name+status+finish reason, latencies, estimated cost, errors) from the agent run summary and per-chat usage hooks. Each signal honors its own `OTEL_*_EXPORTER=none` kill switch, the global `OTEL_SDK_DISABLED`, and declines non-`http/protobuf` protocols independently ([#4604](https://github.com/can1357/oh-my-pi/issues/4604)). +- `retry.fallbackChains` wildcards now support id-prefixed targets and keys: a chain entry like `"openrouter/google/*"` re-prefixes the failing model's bare id (`google-antigravity/gemini-x` → `openrouter/google/gemini-x`), a plain `"provider/*"` entry falling back *from* an aggregator strips the vendor prefix when the target provider only knows the bare id (`openrouter/google/x` → `google-vertex/x`), and an id-prefixed key (`"openrouter/google/*"`) scopes a chain to that provider's ids under the prefix. +- The session tree selector (`/tree`, `/branch`) now supports Shift+Enter to summarize-and-switch in one step: it forks from the selected entry with a branch summary, with no extra prompt and regardless of `branchSummary.enabled`. Plain Enter keeps the current behavior (direct switch by default; the summary prompt only when `branchSummary.enabled` is on). ([#5152](https://github.com/can1357/oh-my-pi/issues/5152)) +- Added the turn's local timestamp (`YYYY-MM-DD HH:mm:ss`, down to the second) to the per-turn token-usage row shown under assistant messages when `display.showTokenUsage` is enabled. + +### Changed + +- Reduced concurrent subagent update CPU by reconstructing recent output only at progress emission boundaries. ([#5936](https://github.com/can1357/oh-my-pi/issues/5936)) +- Fixed `docs/advisor-watchdog.md` overstating advisor delivery for a normal yield: the severity table listed `concern` as unconditionally interrupting and the prose promised a self-ended run could always be steered/resumed. Documented the #4840 terminal-answer exception (`concern` becomes a passive card while `blocker` normally steers, #5628) plus the plan-mode and deferred-ACP constraints that preserve would-be steers until the user resumes ([#5913](https://github.com/can1357/oh-my-pi/issues/5913)). +- Fixed subagent (task) sessions triggering an unnecessary tiny-model session-title generation call on `todo init`. Subagent sessions in a non-interactive host (print/RPC/ACP/eval/SDK/CI) have no operator-visible title and now skip the replan title refresh; interactive hosts keep it, since a live subagent focused from the Agent Hub renders its session name in the status line ([#5910](https://github.com/can1357/oh-my-pi/issues/5910)). +- Fixed `tab.scroll()` timing out after a queued wheel event waits too long for a busy renderer's acknowledgement ([#5905](https://github.com/can1357/oh-my-pi/issues/5905)). +- Made the hashline seen-line guard opt-in and off by default (see `edit.enforceSeenLines`), and stopped excluding column-clipped (>512-char) lines from a snapshot's seen set: a displayed line now counts as seen even when its display was column-truncated, so single-line edits on long lines found via `read`/`grep` apply without a separate full-width re-read. +- Changed the default `astGrep.enabled` setting to `false` +- Batched todo operations with real tool calls to prevent solo todo turns and extra round trips +- Changed every bundled TTSR rule to warn without interrupting generation. +- Renamed the system prompt's project-context section wrapper from `` to `` to stop it colliding with the `task` tool's `context` parameter under in-band XML tool dialects: models were closing `` with a stray `` (primed by the ambient section tag) and emitting sibling params as bare `` elements, so `tasks` arrived missing. +- Rendered `read xd://` calls in the compact grouped read view instead of a full tool-execution card; other internal URLs (`skill://`, `agent://`, …) still render full so their resolved content stays visible. + +### Fixed + +- Fixed the interactive `!`/`!!` shell shortcut spawning fish as a login shell (`fish -l -c …`), which fired `status is-login` blocks in user config (agent/keychain setup, PATH mutation) on every command. fish is now started with `-i` instead — interactive shells source the same `config.fish`/`conf.d` files (so aliases and functions from #1816 keep working) without login-shell side effects. zsh behavior (`-l -i`) is unchanged. +- Fixed the status-line `tok/s` badge ignoring vibe worker sessions: in `/vibe` mode the director is often idle while workers stream, so the badge showed a stale/zero rate while parallel work was actively generating tokens. The rate now aggregates the main session's live tok/s with every live vibe worker's tok/s, and falls back to the main session's own cached rate when no workers are streaming. +- Fixed `plan.defaultOnStartup` being ignored by headless `omp -p` sessions, so the initial prompt now runs in plan mode and the persisted session remains in plan mode for later review ([#6017](https://github.com/can1357/oh-my-pi/issues/6017)). +- Fixed resuming an active plan session replacing its journal-restored model with the current `modelRoles.plan` setting ([#6015](https://github.com/can1357/oh-my-pi/issues/6015)). +- Fixed `--model ` resolving a bare configured `modelRoles` key. +- Browser tool selectors now accept bare snapshot refs (`tab.click("e501")`, `@e501`) everywhere `aria-ref=e501` works — previously the tab-worker backend fell through to a CSS tag selector that could never match, burning the 2s zero-match watchdog with a misleading "matches no elements" hint. `tab.select`, `tab.uploadFile`, `tab.press({ selector })`, `tab.screenshot({ selector })`, and `tab.drag` now resolve refs too. Unknown/stale refs fail immediately with the "refresh refs" error. +- `tab.select` no longer double-reports the previously selected option of a single `` — use `tab.select`. - `tab.waitForNavigation` must start BEFORE the trigger click. - - Navigation invalidates element ids — re-observe. + - Navigation and re-renders (virtualized lists, SPA updates) invalidate ids/refs — re-observe or re-snapshot, then act in the same cell. - Stalled actions fail fast with named error, never whole-cell timeout. + - Raw request interception is run-scoped: run end removes `request` handlers, disables interception, releases held requests. - `app.path` → NEVER tamper with a real desktop app (no stealth patches). - Selectors: CSS + puppeteer `aria/…`, `text/…`, `xpath/…`, `pierce/…`. Playwright-only pseudos (`:has-text()`, `:visible`) are REJECTED. diff --git a/packages/coding-agent/src/prompts/tools/debug.md b/packages/coding-agent/src/prompts/tools/debug.md index b39b65940..d65469d89 100644 --- a/packages/coding-agent/src/prompts/tools/debug.md +++ b/packages/coding-agent/src/prompts/tools/debug.md @@ -1,8 +1,3 @@ Debugger access. Prefer over bash for program state, breakpoints, stepping, or thread inspection. - -Only one active session at a time. `program` is a target path, not a shell command. Directories need a directory-capable adapter (`dlv`). - -Adapters: -- Python: `debugpy` (`pip install debugpy`) -- Go: Delve (`go install github.com/go-delve/delve/cmd/dlv@latest`) -- Ruby: `rdbg` (`gem install debug`) +Only one active session at a time. `program` is a target path, not a shell command. +Directories need a directory-capable adapter (e.g. `dlv`). diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 8ba73935c..765fcef45 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -17,9 +17,12 @@ write(path, content) → str env(key?=None, value?=None) → str | None | dict output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | dict | list[dict] tool.(args) → unknown + Invoke any session tool; `args` = its parameter object. completion(prompt, model?="default"|"smol"|"slow", system?=None, schema?=None) → str | dict -{{#if spawns}}agent(prompt, agent?="{{spawnDefaultAgent}}", model?=None, schema?=None, handle?=False) → str | dict{{#if spawnAllowedAgentsText}} Allowed: {{spawnAllowedAgentsText}}.{{/if}} -{{#if js}} JS: agent(prompt, { agent, schema, handle }).{{/if}} + Oneshot, stateless (no history/tools). `model`: "smol" fast | "default" session | "slow" most capable. `schema` (JSON-Schema) → parsed object. +{{#if spawns}}agent(prompt, agent?="{{spawnDefaultAgent}}", model?=None, label?=None, schema?=None, schema{{#if js}}Mode{{else}}_mode{{/if}}?="permissive", isolated?=None, apply?=None, merge?=None, handle?=False) → str | dict + Run a subagent → final output. `agent` selects a discovered agent; omit it to use `{{spawnDefaultAgent}}`.{{#if spawnAllowedAgentsText}} Allowed agents: {{spawnAllowedAgentsText}}.{{/if}} `schema` overrides agent/session schemas; `schemaMode`/`schema_mode`: "permissive" | "strict". Effective schemas return parsed data. `isolated` requests a worktree; `apply`/`merge` control its changes. Background via `local://` files named in the prompt. `handle` → { text, output, handle: "agent://", id, agent }, parsed `data` when structured. +{{#if js}} JS: ONE trailing object — agent(prompt, { agent, model, label, schema, schemaMode, isolated, apply, merge, handle }).{{/if}} {{/if}} parallel(thunks) → list pipeline(items, ...stages) → list log(message) → None phase(title) → None diff --git a/packages/coding-agent/src/prompts/tools/github.md b/packages/coding-agent/src/prompts/tools/github.md index 3997cd105..1056d8ec8 100644 --- a/packages/coding-agent/src/prompts/tools/github.md +++ b/packages/coding-agent/src/prompts/tools/github.md @@ -1,8 +1,9 @@ -Op-based `gh` wrapper: repos, PRs, search, checkout, push, Actions watch. Read an issue/PR via `issue://`/`pr://`. PR diffs: `pr:///diff` (file listing), `pr:///diff/` (file slice, 1-indexed), `pr:///diff/all` (full diff). +Op-based `gh` wrapper: repos, repository files, PRs, search, checkout, push, Actions watch. Read an issue/PR via `issue://`/`pr://`. PR diffs: `pr:///diff` (file listing), `pr:///diff/` (file slice, 1-indexed), `pr:///diff/all` (full diff). Pick op via `op`. Beyond the field descriptions, per op: - `repo_view` — omit `repo` to view the current checkout. +- `file_read` — reads `path` from `repo`; omit `repo` for the current checkout and `branch` for its default branch. - `pr_create` — `head` defaults to the current branch. - `pr_checkout` — checks PR(s) out into dedicated git worktrees, not your working tree; pass an array of `pr` to batch multiple in one call. - `pr_push` — requires the branch to have been checked out first via `op: pr_checkout`. @@ -15,3 +16,7 @@ Pick op via `op`. Beyond the field descriptions, per op: Concise summary per op. `run_watch` failures save full logs to a session artifact. + + +GitHub-hosted repository file? MUST use `file_read`; NEVER `curl`/`wget`. + diff --git a/packages/coding-agent/src/prompts/tools/hub.md b/packages/coding-agent/src/prompts/tools/hub.md index 73be57b32..0d48b70e9 100644 --- a/packages/coding-agent/src/prompts/tools/hub.md +++ b/packages/coding-agent/src/prompts/tools/hub.md @@ -3,7 +3,7 @@ Use `op: "list"` to discover peers. Address peers by exact roster ID — NEVER i # Messaging & Jobs -Background jobs deliver their results automatically the moment they finish. You NEVER need to poll for output — intervene only to block, kill, or inspect. +Background jobs auto-deliver when they finish. You NEVER need to poll; if `jobs`/`wait` observes a settled job first, that snapshot is the delivery and suppresses duplicate `async-result`. - **`send`** (with `to`): fire-and-forget, NEVER blocks. Delivery receipts (`delivered`/`failed`) immediate; `failed` → peer gone, don't retry. Sending wakes `idle`/`parked` peers. Answering: lead with answer, NEVER quote, set `replyTo`. @@ -12,7 +12,9 @@ Background jobs deliver their results automatically the moment they finish. You - Bare `wait` watches every running job AND incoming messages. NEVER pass an array of every running ID; `ids` narrows to specific jobs, `from` to one peer (or use `await: true` on send). - **`inbox`**: drain queued messages without blocking. - **`cancel`**: kill background jobs by `ids` when they have hung, stalled, or are no longer needed. Returns immediately. -- **`jobs`**: status snapshot of every job without waiting. Also names running subagents with no job entry — coordinate with those via `send`. +- **`jobs`**: status snapshot of every job without waiting. A settled row consumes auto-delivery. Also names running subagents with no job entry — coordinate with those via `send`. +- Job rows are process-local and expire roughly five minutes after settlement. Afterward, use the agent ID with `send`, `agent://`, or `history://`. +- `completed` means successful yield/job exit, not artifact acceptance. Verify claimed changes. - NEVER use shell tools, grep, or read other sessions' files to figure out what a peer is doing. Message them directly. - NEVER use hub messaging for something a tool can answer (e.g., grepping codebase, running a build). diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md index b2bcd639c..47610ccdb 100644 --- a/packages/coding-agent/src/prompts/tools/read.md +++ b/packages/coding-agent/src/prompts/tools/read.md @@ -15,7 +15,7 @@ Read files, directories, archives, SQLite, images, documents, internal resources - {{#if IS_HL_MODE}}File + selector → `[foo.ts#1A2B]` snapshot header + numbered lines. Copy `[FILENAME#TAG]` for anchored edits; NEVER fabricate the tag.{{/if}} - Directory → depth-limited dirent listing. - SQLite (`.sqlite`, `.sqlite3`, `.db`, `.db3`): `file.db` (tables), `file.db:table` (schema+rows), `file.db:table:key` (by PK), `?limit=`/`?where=`/`?q=SELECT`. -- Archives (`.tar`, `.tar.gz`, `.tgz`, `.zip`): `archive.ext:path/inside/archive` reads a member. +- Archives (`.tar`, `.tar.gz`, `.tgz`, `.zip`, plus ZIP-based `.jar`/`.war`/`.ear`/`.apk`): `archive.ext:path/inside/archive` reads a member. - Documents → extracted text. Notebooks → editable cells. Images → {{#if INSPECT_IMAGE_ENABLED}}metadata; call `inspect_image`{{else}}decoded inline{{/if}}. `:raw` bypasses converters. - URLs → reader-mode clean text/markdown; `:raw` → untouched HTML. Bare `host:port` needs trailing slash. - Internal URIs — all schemes take selectors. `artifact://` recovers spilled output; page with `:N-M`/`:raw:N-M`. diff --git a/packages/coding-agent/src/prompts/tools/task-async-contract.md b/packages/coding-agent/src/prompts/tools/task-async-contract.md new file mode 100644 index 000000000..95d6efeae --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/task-async-contract.md @@ -0,0 +1 @@ +No polling is needed. Inspecting a settled job with `hub jobs` or `hub wait` makes that snapshot its delivery, so no duplicate `async-result` follows. Job IDs live in process memory for roughly five minutes after settlement; afterward, use the agent ID with `hub send`, `agent://`, or `history://`. `completed` means the subagent yielded successfully, not that claimed artifacts were verified. diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md index d41cf71e3..0e3be887f 100644 --- a/packages/coding-agent/src/prompts/tools/task.md +++ b/packages/coding-agent/src/prompts/tools/task.md @@ -1,7 +1,14 @@ {{#if asyncEnabled}}{{#if batchEnabled}}Delegate work to background subagents by passing multiple items in a single `tasks[]` batch. -Execution does not block — you receive IDs immediately; results deliver when subagents finish.{{else}}Delegate work to ONE background subagent per call. -Execution does not block — you receive an ID immediately; the result delivers when the subagent finishes.{{/if}}{{#if hasBlockingAgents}} +Execution does not block — you receive IDs immediately.{{else}}Delegate work to ONE background subagent per call. +Execution does not block — you receive an ID immediately.{{/if}}{{#if hasBlockingAgents}} Agents marked BLOCKING run inline — results return in this call; non-blocking items in the same batch still spawn as background jobs.{{/if}}{{else}}{{#if batchEnabled}}Run subagents synchronously by passing items in a `tasks[]` batch. Execution blocks until all work finishes.{{else}}Run ONE subagent synchronously. Execution blocks until work finishes.{{/if}}{{/if}} +{{#if asyncEnabled}} + +# Async Job Contract +- Results auto-deliver. A settled `hub jobs`/`hub wait` snapshot is the delivery; no duplicate `async-result` follows. +- Job IDs are process-local and expire roughly five minutes after settlement. Afterward, use the agent ID with `hub send`, `agent://`, or `history://`. +- `completed` means successful yield/job exit, not artifact acceptance. Verify claimed changes. +{{/if}} # Task Design - **Agent typing:** Pick each item's `agent` type. Read-only research MUST use `agent: "scout"` (faster model). Use default worker only when no specialist fits. @@ -10,20 +17,34 @@ Agents marked BLOCKING run inline — results return in this call; non-blocking # Inputs {{#if batchEnabled}} -- `context`: Shared project state for the entire batch — don't duplicate into individual tasks. -- `tasks[]`: Subagents to spawn. - - `name`: CamelCase ≤32 chars (auto-generated if omitted). - - `agent`: specialist type (optional). - - `task`: Complete, self-contained instructions — no one-liners, no missing acceptance criteria. +- `context`: Shared project state, constraints, and contracts. Applies to the entire batch; do not duplicate this background into individual tasks. +- `tasks[]`: Array of subagents to spawn. + - `name`: A stable CamelCase identifier (≤32 chars), used to address the agent (IRC, job ids). Generated automatically if omitted. + - `agent`: The agent type running this item (e.g. `scout`, `reviewer`). Omitting it gives you the general-purpose worker (`{{defaultAgent}}`) — NEVER pass that name explicitly. Only omit it after checking the agent list below and finding no specialist that fits.{{#if allowedAgentsText}} Current spawn policy allows: {{allowedAgentsText}}.{{/if}} + - `task`: Complete, self-contained instructions. One-liners or missing acceptance criteria are PROHIBITED. + - `model`: Explicit non-empty model selector or non-empty fallback chain for this spawn. A `:reasoning` suffix is preserved. Overrides agent-specific model settings. + - `outputSchema`: Invocation-specific JSON Schema. Overrides the selected agent and parent-session schemas. + - `schemaMode`: `"permissive"` (default) accepts a retry-exhausted invalid result with a warning; `"strict"` fails it. {{#if isolationEnabled}} - - `isolated`: Run in dedicated worktree, return patches. Destroyed on completion, cannot be addressed afterward. +{{#if applyIsolatedChanges}} + - `isolated`: Run in a dedicated worktree; successful changes are automatically applied to the parent checkout. +{{else}} + - `isolated`: Run in a dedicated worktree; changes are retained as patch or branch artifacts without modifying the parent checkout. +{{/if}} {{/if}} {{else}} -- `name`: CamelCase ≤32 chars (auto-generated if omitted). -- `agent`: specialist type (optional). -- `task`: Complete, self-contained instructions — no one-liners, no missing acceptance criteria. +- `name`: A stable CamelCase identifier (≤32 chars), used to address the agent (IRC, job ids). Generated automatically if omitted. +- `agent`: The agent type to spawn (e.g. `scout`, `reviewer`). Omitting it gives you the general-purpose worker (`{{defaultAgent}}`) — NEVER pass that name explicitly. Only omit it after checking the agent list below and finding no specialist that fits.{{#if allowedAgentsText}} Current spawn policy allows: {{allowedAgentsText}}.{{/if}} +- `task`: Complete, self-contained instructions. One-liners or missing acceptance criteria are PROHIBITED. +- `model`: Explicit non-empty model selector or non-empty fallback chain for this spawn. A `:reasoning` suffix is preserved. Overrides agent-specific model settings. +- `outputSchema`: Invocation-specific JSON Schema. Overrides the selected agent and parent-session schemas. +- `schemaMode`: `"permissive"` (default) accepts a retry-exhausted invalid result with a warning; `"strict"` fails it. {{#if isolationEnabled}} -- `isolated`: Run in dedicated worktree, return patches. +{{#if applyIsolatedChanges}} +- `isolated`: Run in a dedicated worktree; successful changes are automatically applied to the parent checkout. +{{else}} +- `isolated`: Run in a dedicated worktree; changes are retained as patch or branch artifacts without modifying the parent checkout. +{{/if}} {{/if}} {{/if}} diff --git a/packages/coding-agent/src/prompts/tools/write.md b/packages/coding-agent/src/prompts/tools/write.md index 00af5c3a6..d03765c43 100644 --- a/packages/coding-agent/src/prompts/tools/write.md +++ b/packages/coding-agent/src/prompts/tools/write.md @@ -3,7 +3,7 @@ Creates or overwrites file at specified path. - Creating new files explicitly required by task - Replacing entire file contents when editing would be more complex -- Supports `.tar`, `.tar.gz`, `.tgz`, and `.zip` archive entries via `archive.ext:path/inside/archive` +- Supports `.tar`, `.tar.gz`, `.tgz`, `.zip`, and ZIP-based `.jar`/`.war`/`.ear`/`.apk` archive entries via `archive.ext:path/inside/archive` - Supports SQLite row operations via `db.sqlite:table` (insert), `db.sqlite:table:key` (update with JSON content, delete with empty content) diff --git a/packages/coding-agent/src/registry/agent-lifecycle.ts b/packages/coding-agent/src/registry/agent-lifecycle.ts index 91678e2a7..fb79eb274 100644 --- a/packages/coding-agent/src/registry/agent-lifecycle.ts +++ b/packages/coding-agent/src/registry/agent-lifecycle.ts @@ -8,6 +8,12 @@ * sessionFile), and revives it on demand through * {@link AgentLifecycleManager.ensureLive}. Only this manager flips * `parked` ↔ `idle`. + * + * Park/dispose is gated against concurrent ensureLive/hub-send: + * - A disposing session is never handed out. + * - ensureLive during an in-flight park either cancels the park (session still + * live) or waits for detach+park and then revives. + * - Concurrent ensureLive/park operations coalesce per id. */ import { logger } from "@oh-my-pi/pi-utils"; @@ -38,6 +44,17 @@ interface AdoptedAgent { timer?: NodeJS.Timeout; } +interface ParkInFlight { + /** Resolves when the park attempt finishes (success, cancel, or dispose error). */ + promise: Promise; + /** Cancel before the session is detached. Returns true if cancel took effect. */ + cancel: () => boolean; + /** True once cancel() succeeded (ensureLive kept the live session). */ + cancelled: boolean; + /** True once the live session has been detached and status is parked. */ + detached: boolean; +} + export class AgentLifecycleManager { static #global: AgentLifecycleManager | undefined; @@ -59,7 +76,7 @@ export class AgentLifecycleManager { } current.#adopted.clear(); current.#revivals.clear(); - current.#parking.clear(); + current.#parks.clear(); current.#persistedReviverFactory = undefined; } AgentLifecycleManager.#global = undefined; @@ -67,8 +84,11 @@ export class AgentLifecycleManager { readonly #registry: AgentRegistry; readonly #adopted = new Map(); - /** Ids whose session is being disposed by {@link park} right now. */ - readonly #parking = new Set(); + /** + * In-flight park attempts. A park is cancelable until the live session is + * detached; after detach, ensureLive waits for the park and revives. + */ + readonly #parks = new Map(); /** In-flight revives, so concurrent {@link ensureLive} calls coalesce. */ readonly #revivals = new Map>(); #unsubscribe: (() => void) | undefined; @@ -114,44 +134,129 @@ export class AgentLifecycleManager { return this.#adopted.has(id); } - /** True while {@link park} is disposing this agent's session (lets dispose hooks distinguish park from teardown). */ + /** + * True when this manager owns `registry` — i.e. its adopt/park/revive state + * describes that registry's refs. Lets a caller holding a specific registry + * (e.g. a custom-registry {@link IrcBus} that fell back to the global + * manager) skip lifecycle gating that would consult unrelated park state. + */ + manages(registry: AgentRegistry): boolean { + return this.#registry === registry; + } + + /** + * True while {@link park} is disposing this agent's session (lets dispose + * hooks distinguish park from teardown). False once the park is cancelled + * by ensureLive or after detach+dispose completes. + */ isParking(id: string): boolean { - return this.#parking.has(id); + const park = this.#parks.get(id); + return Boolean(park && !park.cancelled); } /** * Dispose the live session, detach it from the registry, and mark the * agent `parked`. No-op unless the id is adopted and live. + * + * The session is detached (and status flipped to `parked`) *before* + * `session.dispose()` so concurrent {@link ensureLive}/hub-send never + * observe or inject into a disposing session. A concurrent ensureLive that + * arrives before detach cancels the park and keeps the live session. */ async park(id: string): Promise { + const existing = this.#parks.get(id); + if (existing) return existing.promise; + const adopted = this.#adopted.get(id); if (!adopted) return; const ref = this.#registry.get(id); - if (!ref?.session) return; + const session = ref?.session; + if (!session) return; + if (adopted.timer) { clearTimeout(adopted.timer); adopted.timer = undefined; } - this.#parking.add(id); - try { + + let cancelled = false; + const park: ParkInFlight = { + promise: undefined as unknown as Promise, + cancel: () => { + // Cancel only before detach — once detached the old session is already + // leaving the registry and must finish disposing. + if (park.detached || cancelled) return cancelled; + cancelled = true; + park.cancelled = true; + return true; + }, + cancelled: false, + detached: false, + }; + + park.promise = (async () => { try { - await ref.session.dispose(); - } catch (error) { - logger.warn("AgentLifecycleManager.park: session dispose failed", { id, error: String(error) }); + // Yield so a same-tick ensureLive/hub-send can cancel before we + // commit to dispose. Deterministic with Promise microtasks; no timers. + await Promise.resolve(); + if (cancelled) return; + + // Re-check liveness: release/unregister may have raced us. + const live = this.#registry.get(id); + if (!live?.session || live.session !== session) return; + if (!this.#adopted.has(id)) return; + + // Commit: detach + parked *before* dispose so callers never see a + // dying session via ref.session / idle status. + park.detached = true; + this.#registry.detachSession(id); + this.#registry.setStatus(id, "parked"); + + try { + await session.dispose(); + } catch (error) { + logger.warn("AgentLifecycleManager.park: session dispose failed", { id, error: String(error) }); + } + } finally { + // Only clear if we are still the in-flight entry (a later park would + // have replaced us only after we resolved). + if (this.#parks.get(id) === park) this.#parks.delete(id); } - this.#registry.detachSession(id); - this.#registry.setStatus(id, "parked"); - } finally { - this.#parking.delete(id); - } + })(); + + this.#parks.set(id, park); + return park.promise; } /** * Return the live session, reviving from the sessionFile if parked. * Throws a plain Error if the id is unknown or parked without a reviver. * Concurrent calls share one in-flight revive. + * + * Never returns a session that is mid-dispose: an in-flight park is either + * cancelled (session still live) or awaited to completion before revive. */ async ensureLive(id: string): Promise { + const park = this.#parks.get(id); + if (park) { + const ref = this.#registry.get(id); + // Cancel if the live session is still attached — keep it instead of + // thrashing dispose + revive. + if (ref?.session && !park.detached && park.cancel()) { + await park.promise; + const kept = this.#registry.get(id)?.session; + if (kept) { + // Park cleared the idle timer; re-arm so TTL park still works. + const adopted = this.#adopted.get(id); + if (adopted && ref.status === "idle") this.#armTimer(id, adopted); + return kept; + } + } else { + // Already committed to detach (or no live session): wait for park, + // then fall through to the revive path. + await park.promise; + } + } + const ref = this.#registry.get(id); if (!ref) { throw new Error( @@ -208,6 +313,14 @@ export class AgentLifecycleManager { const adopted = this.#adopted.get(id); clearTimeout(adopted?.timer); this.#adopted.delete(id); + + const park = this.#parks.get(id); + if (park) { + // Prefer cancel when the session is still live so release owns dispose. + if (!park.detached) park.cancel(); + await park.promise; + } + const ref = this.#registry.get(id); if (ref?.session) { try { @@ -223,10 +336,10 @@ export class AgentLifecycleManager { async dispose(): Promise { this.#unsubscribe?.(); this.#unsubscribe = undefined; - const ids = [...this.#adopted.keys()]; + const ids = [...new Set([...this.#adopted.keys(), ...this.#parks.keys()])]; await Promise.all(ids.map(id => this.release(id))); this.#revivals.clear(); - this.#parking.clear(); + this.#parks.clear(); this.#persistedReviverFactory = undefined; } @@ -264,6 +377,8 @@ export class AgentLifecycleManager { adopted.timer = undefined; } } else if (event.ref.status === "idle") { + // Don't re-arm while a park is in flight — the park owns the transition. + if (this.#parks.has(event.ref.id)) return; this.#armTimer(event.ref.id, adopted); } } diff --git a/packages/coding-agent/src/registry/persisted-agents.ts b/packages/coding-agent/src/registry/persisted-agents.ts new file mode 100644 index 000000000..91fe4008c --- /dev/null +++ b/packages/coding-agent/src/registry/persisted-agents.ts @@ -0,0 +1,74 @@ +import * as fs from "node:fs"; +import * as path from "node:path"; +import { ADVISOR_TRANSCRIPT_FILENAME, isAdvisorTranscriptName } from "../advisor/transcript-recorder"; +import { type AgentRegistry, MAIN_AGENT_ID } from "./agent-registry"; + +/** Register persisted subagent and advisor transcripts as parked registry refs. */ +export async function registerPersistedSubagents( + registry: AgentRegistry, + sessionFile: string | null | undefined, +): Promise { + if (!sessionFile?.endsWith(".jsonl")) return; + const root = sessionFile.slice(0, -6); + await registerPersistedSubagentsFromDir(registry, root, undefined); +} + +async function registerPersistedSubagentsFromDir( + registry: AgentRegistry, + dir: string, + parentId: string | undefined, +): Promise { + let entries: fs.Dirent[]; + try { + entries = await fs.promises.readdir(dir, { withFileTypes: true }); + } catch { + return; + } + for (const entry of entries) { + if (!entry.isFile() || !entry.name.endsWith(".jsonl") || entry.name.includes(".bak")) continue; + const sessionFile = path.join(dir, entry.name); + // The advisor transcript is observability-only: register it as a non-peer + // `advisor` kind under its owning session so the Hub can show its read-only + // transcript, but it never joins agent-facing rosters and is not revivable. + if (isAdvisorTranscriptName(entry.name)) { + const owner = parentId ?? MAIN_AGENT_ID; + // `__advisor.jsonl` → the default advisor (no slug); `__advisor..jsonl` + // → a named advisor, keyed and labeled by its slug. + const slug = + entry.name === ADVISOR_TRANSCRIPT_FILENAME ? "" : entry.name.slice("__advisor.".length, -".jsonl".length); + const advisorId = slug ? `${owner}/advisor:${slug}` : `${owner}/advisor`; + const displayName = slug ? `advisor:${slug}` : "advisor"; + const existing = registry.get(advisorId); + // Never clobber a non-advisor ref that happens to share this id (a freak + // user task literally named `/advisor`): leave it, skip the advisor. + if (existing && existing.kind !== "advisor") continue; + if (existing?.sessionFile !== sessionFile) { + // The id is reused across `/new`; refresh it to the current session's file. + if (existing) registry.unregister(advisorId); + registry.register({ + id: advisorId, + displayName, + kind: "advisor", + parentId: owner, + session: null, + sessionFile, + status: "parked", + }); + } + continue; + } + const id = entry.name.slice(0, -6); + if (!registry.get(id)) { + registry.register({ + id, + displayName: id, + kind: "sub", + parentId: parentId ?? MAIN_AGENT_ID, + session: null, + sessionFile, + status: "parked", + }); + } + await registerPersistedSubagentsFromDir(registry, path.join(dir, id), id); + } +} diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 788c51adc..833359c11 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -2,13 +2,21 @@ import { Agent, type AgentEvent, type AgentMessage, + type AgentOptions, type AgentTelemetryConfig, type AgentTool, AppendOnlyContextManager, filterProviderReplayMessages, type ThinkingLevel, } from "@oh-my-pi/pi-agent-core"; -import type { Context, CredentialDisabledEvent, Message, Model, SimpleStreamOptions } from "@oh-my-pi/pi-ai"; +import type { + Context, + CredentialDisabledEvent, + Message, + Model, + ProviderSessionState, + SimpleStreamOptions, +} from "@oh-my-pi/pi-ai"; import type { Dialect } from "@oh-my-pi/pi-ai/dialect"; import { getOpenAICodexTransportDetails, @@ -26,6 +34,7 @@ import { } from "./advisor"; import { type AsyncJob, AsyncJobManager } from "./async"; import { AutoLearnController, buildAutoLearnInstructions } from "./autolearn/controller"; +import { createAutoresearchExtension } from "./autoresearch"; import { loadCapability } from "./capability"; import { type Rule, ruleCapability, setActiveRules } from "./capability/rule"; import { bucketRules } from "./capability/rule-buckets"; @@ -41,6 +50,7 @@ import { parseModelString, pickDefaultAvailableModel, resolveAllowedModels, + resolveCliModel, resolveConfiguredModelPatterns, resolveModelRoleValue, } from "./config/model-resolver"; @@ -109,7 +119,7 @@ import { obfuscateProviderContext, SecretObfuscator, } from "./secrets"; -import { AgentSession, type PlanYolo, type Prewalk } from "./session/agent-session"; +import { AgentSession, type InitialRetryFallbackState, type PlanYolo, type Prewalk } from "./session/agent-session"; import { discoverAuthStorage as discoverAuthStorageFromConfig } from "./session/auth-broker-config"; import type { AuthStorage } from "./session/auth-storage"; import { createInterruptedTurnAbortMessage } from "./session/exit-diagnostics"; @@ -137,6 +147,7 @@ import { } from "./system-prompt"; import { AgentOutputManager } from "./task/output-manager"; import { wrapStreamFnWithProviderConcurrency } from "./task/provider-concurrency"; +import type { StructuredSubagentSchemaMode } from "./task/types"; import { AUTO_THINKING, type ConfiguredThinkingLevel, @@ -154,6 +165,7 @@ import { createTools, createVibeTools, type DeferredDiagnosticsEntry, + defaultLoadModeForToolName, discoverStartupLspServers, EditTool, EvalTool, @@ -378,6 +390,8 @@ export interface CreateAgentSessionOptions { modelPatternAuthFallback?: string; /** Role name used to install retry fallbacks after deferred subagent patterns resolve. */ modelPatternFallbackRole?: string; + /** Validated default retry chain to install when a deferred singleton pattern resolves. */ + modelPatternDefaultFallbackChain?: string[]; /** Thinking selector. Default: from settings, else unset */ thinkingLevel?: ConfiguredThinkingLevel; /** Models available for cycling (Ctrl+P in interactive mode) */ @@ -473,20 +487,30 @@ export interface CreateAgentSessionOptions { /** File-based slash commands. Default: discovered from commands/ directories */ slashCommands?: FileSlashCommand[]; - /** Enable MCP server discovery from .mcp.json files. Default: true */ + /** + * Enable MCP capabilities. `false` skips MCP discovery and ignores + * `mcpManager`, preventing process-global or inherited MCP access. Default: + * true. + */ enableMCP?: boolean; - /** Existing MCP manager to reuse (skips discovery, propagates to toolSession). */ + /** Existing MCP manager to reuse when MCP is enabled (skips discovery, propagates to toolSession). */ mcpManager?: MCPManager; /** Enable LSP integration (tool, formatting, diagnostics, warmup). Default: true */ enableLsp?: boolean; + /** Whether this invocation may expose IRC. `false` removes it even for subagents. */ + enableIrc?: boolean; /** Skip subprocess-kernel availability checks and prelude warmup */ skipPythonPreflight?: boolean; /** Tool names explicitly requested (enables disabled-by-default tools) */ toolNames?: string[]; + /** Limit the session to explicitly supplied tool names, without discovered extras. */ + restrictToolNames?: boolean; - /** Output schema for structured completion (subagents) */ + /** Output schema for structured completion (subagents). */ outputSchema?: unknown; + /** Enforcement policy for {@link outputSchema}; defaults to legacy permissive behavior. */ + outputSchemaMode?: StructuredSubagentSchemaMode; /** Whether to include the yield tool by default */ requireYieldTool?: boolean; /** Task recursion depth (for subagent sessions). Default: 0 */ @@ -883,16 +907,19 @@ function registerEvalCleanup(): void { postmortem.register("julia-cleanup", disposeAllJuliaKernelSessions); } -function customToolToDefinition(tool: CustomTool): ToolDefinition { +export function customToolToDefinition(tool: CustomTool): ToolDefinition { const definition: ToolDefinition & { [TOOL_DEFINITION_MARKER]: true } = { name: tool.name, label: tool.label, description: tool.description, parameters: tool.parameters, hidden: tool.hidden, - loadMode: tool.loadMode ?? "discoverable", + loadMode: defaultLoadModeForToolName(tool.name, tool.loadMode), deferrable: tool.deferrable, approval: typeof tool.approval === "function" ? tool.approval.bind(tool) : tool.approval, + // Preserved through RegisteredToolAdapter so MCP-backed tools' explicit + // `strict: false` (#4336/#4340) survives the custom-tool → definition bridge. + strict: tool.strict, mcpServerName: tool.mcpServerName, mcpToolName: tool.mcpToolName, execute: (toolCallId, params, signal, onUpdate, ctx) => @@ -1055,6 +1082,88 @@ function buildMCPPromptCommands(manager: MCPManager): LoadedCustomCommand[] { } return commands; } + +/** Dependencies used to construct an isolated auto-learn capture agent. */ +export interface AutoLearnCaptureRunnerOptions { + sourceAgent: Agent; + captureTools: AgentTool[]; + createAgent: (options: AgentOptions) => Agent; + onPayload?: SimpleStreamOptions["onPayload"]; + onResponse?: SimpleStreamOptions["onResponse"]; + createSessionId?: () => string; +} + +/** Build a private capture runner over a detached message snapshot and provider session. */ +export function createAutoLearnCaptureRunner( + options: AutoLearnCaptureRunnerOptions, +): (content: string, signal?: AbortSignal) => Promise { + return async (content, signal) => { + if (options.captureTools.length === 0 || signal?.aborted) return; + const captureModel = options.sourceAgent.state.model; + if (!captureModel) return; + + const captureSessionId = options.createSessionId?.() ?? Bun.randomUUIDv7(); + const captureProviderSessionState = new Map(); + const captureMessages = options.sourceAgent.state.messages.map((message): AgentMessage => { + if (message.role === "assistant") { + return { ...message, responseId: undefined, providerPayload: undefined }; + } + if (message.role === "user" || message.role === "developer") { + return { ...message, providerPayload: undefined }; + } + return message; + }); + const captureAgent = options.createAgent({ + initialState: { + systemPrompt: [...options.sourceAgent.state.systemPrompt], + model: captureModel, + thinkingLevel: options.sourceAgent.state.thinkingLevel, + disableReasoning: options.sourceAgent.state.disableReasoning, + tools: options.captureTools, + messages: captureMessages, + }, + sessionId: captureSessionId, + promptCacheKey: captureSessionId, + providerSessionState: captureProviderSessionState, + getApiKey: requestModel => options.sourceAgent.getApiKey?.(requestModel), + onPayload: options.onPayload, + onResponse: options.onResponse, + }); + captureAgent.setMetadataResolver(provider => options.sourceAgent.metadataForProvider(provider)); + const captureMessage: CustomMessage = { + role: "custom", + customType: "autolearn-nudge", + content, + display: false, + attribution: "agent", + timestamp: Date.now(), + }; + const abortCapture = () => captureAgent.abort(signal?.reason); + signal?.addEventListener("abort", abortCapture, { once: true }); + try { + if (signal?.aborted) { + abortCapture(); + return; + } + await captureAgent.prompt(captureMessage); + } catch (error) { + if (!signal?.aborted) throw error; + } finally { + signal?.removeEventListener("abort", abortCapture); + for (const [providerKey, state] of captureProviderSessionState) { + try { + state.close(); + } catch (error) { + logger.warn("Failed to close auto-learn capture provider state", { + providerKey, + error: String(error), + }); + } + } + captureProviderSessionState.clear(); + } + }; +} /** * Create an AgentSession with the specified options. * @@ -1281,6 +1390,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} ); let model = options.model; let modelFallbackMessage: string | undefined; + let initialRetryFallback: InitialRetryFallbackState | undefined; // Identify session model strings to restore in fallback order. We do an // initial pass here so model-dependent setup (thinking-level resolution, // host preconnect) can use the restored model; extension-registered @@ -1453,7 +1563,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} let session!: AgentSession; let hasSession = false; let hasRegistered = false; - const enableLsp = options.enableLsp ?? true; + const restrictToolNames = options.restrictToolNames === true; + const enableLsp = !restrictToolNames && (options.enableLsp ?? true); const asyncMaxJobs = Math.min(100, Math.max(1, settings.get("async.maxJobs") ?? 100)); const ASYNC_INLINE_RESULT_MAX_CHARS = 12_000; const ASYNC_PREVIEW_MAX_CHARS = 4_000; @@ -1550,17 +1661,25 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} setActiveToolNames, hasUI: options.hasUI ?? false, enableLsp, + enableIrc: restrictToolNames ? false : options.enableIrc, + restrictToolNames, get hasEditTool() { const requestedToolNames = options.toolNames ? normalizeToolNames(options.toolNames) : undefined; - return !requestedToolNames || requestedToolNames.includes("edit"); + return restrictToolNames + ? requestedToolNames?.includes("edit") === true + : !requestedToolNames || requestedToolNames.includes("edit"); }, skipPythonPreflight: options.skipPythonPreflight, contextFiles, workspaceTree: resolvedWorkspaceTree, - skills, + get skills() { + return session?.skills ?? skills; + }, + refreshSkills: () => session.refreshSkills(), rules: allRules, eventBus, outputSchema: options.outputSchema, + outputSchemaMode: options.outputSchemaMode, requireYieldTool: options.requireYieldTool, prewalkArmed: options.prewalk !== undefined, taskDepth: options.taskDepth ?? 0, @@ -1577,6 +1696,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} getAgentId: () => resolvedAgentId, getToolByName: name => session?.getToolByName(name), agentRegistry, + // The global lifecycle releases through AgentRegistry.global(); wiring it + // onto a caller-supplied registry would report a cancel while releasing an + // unrelated global ref. With no lifecycle, hub cancel falls back to + // dispose + unregister on the session's own registry. + agentLifecycle: options.agentRegistry ? undefined : () => AgentLifecycleManager.global(), getSessionSpawns: () => options.spawns ?? "*", getModelString: () => (hasExplicitModel && model ? formatModelString(model) : undefined), getActiveModelString, @@ -1677,10 +1801,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Create built-in tools (already wrapped with meta notice formatting) const builtinTools = await logger.time("createAllTools", createTools, toolSession, options.toolNames); - // Discover MCP tools from .mcp.json files - let mcpManager: MCPManager | undefined = options.mcpManager; + // Restricted sessions cannot inherit or discover MCP capabilities. + const enableMCP = !restrictToolNames && (options.enableMCP ?? true); + let mcpManager: MCPManager | undefined = enableMCP ? options.mcpManager : undefined; toolSession.mcpManager = mcpManager; - const enableMCP = options.enableMCP ?? true; + toolSession.enableMCP = enableMCP; const deferMCPDiscoveryForUI = enableMCP && !mcpManager && options.hasUI === true; const customTools: CustomTool[] = []; let startDeferredMCPDiscovery: ((liveSession: AgentSession) => void) | undefined; @@ -1766,58 +1891,62 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // to mirror the AsyncJobManager ownership rule. if (mcpManager && !options.parentTaskPrefix) MCPManager.setInstance(mcpManager); - // Add image tools when generation is enabled and either no explicit tool - // whitelist was given or it names `generate_image`. Image gen is a - // discoverable custom tool: once it enters the registry the common - // partition presents it under xd:// (or routes it to BM25 discovery), so no - // source-specific force-activation is needed — only this eligibility gate. - const imageGenRequested = !options.toolNames || options.toolNames.includes("generate_image"); - if (settings.get("generate_image.enabled") && imageGenRequested) { - const imageGenTools = await logger.time("getImageGenTools", () => getImageGenTools(modelRegistry, model)); - if (imageGenTools.length > 0) { - customTools.push(...(imageGenTools as unknown as CustomTool[])); - } - } - - if (settings.get("speechgen.enabled")) { - customTools.push(ttsTool as unknown as CustomTool); - } - - // Add web search tools - if (options.toolNames?.includes("web_search")) { - customTools.push(...getSearchTools()); - } - - // Discover custom tools from `.omp/tools/`, `.claude/tools/`, plugins, etc. - // Subagents reuse the parent's scan via `preloadedCustomToolPaths` to skip - // the FS walk, but ALWAYS re-call `loadCustomTools` here so factories bind - // to THIS session's `CustomToolAPI` (cwd, exec, pushPendingAction, UI). - // Forwarding the parent's `LoadedCustomTool[]` directly would route tool - // execution back through the parent — wrong for isolated tasks and for - // pending-action queueing. const builtInToolNames = builtinTools.map(t => t.name); - const customToolPaths: ToolPathWithSource[] = - options.preloadedCustomToolPaths ?? - (await logger.time("discoverCustomToolPaths", () => discoverCustomToolPaths([], cwd))); - const customToolsLoadResult = await logger.time("loadCustomTools", () => - loadCustomTools(customToolPaths, cwd, builtInToolNames, action => queueResolveHandler(toolSession, action)), - ); - for (const { path, error } of customToolsLoadResult.errors) { - logger.error("Custom tool load failed", { path, error }); - } - if (customToolsLoadResult.tools.length > 0) { - customTools.push(...customToolsLoadResult.tools.map(loaded => loaded.tool)); + let customToolPaths: ToolPathWithSource[] = []; + const inlineExtensions: ExtensionFactory[] = []; + if (!restrictToolNames) { + // Add image tools when generation is enabled and either no explicit tool + // whitelist was given or it names `generate_image`. Unlike built-in tools + // (filtered in `createTools`), custom tools are force-activated via + // `alwaysInclude` below, so an explicit `--no-tools`/whitelist must be + // honored here or image-gen would leak past every filter (issue #5305). + const imageGenRequested = !options.toolNames || options.toolNames.includes("generate_image"); + if (settings.get("generate_image.enabled") && imageGenRequested) { + const imageGenTools = await logger.time("getImageGenTools", () => getImageGenTools(modelRegistry, model)); + if (imageGenTools.length > 0) { + customTools.push(...(imageGenTools as unknown as CustomTool[])); + } + } + + if (settings.get("speechgen.enabled")) { + customTools.push(ttsTool as unknown as CustomTool); + } + + // Add web search tools + if (options.toolNames?.includes("web_search")) { + customTools.push(...getSearchTools()); + } + + // Discover custom tools from `.omp/tools/`, `.claude/tools/`, plugins, etc. + // Subagents reuse the parent's scan via `preloadedCustomToolPaths` to skip + // the FS walk, but ALWAYS re-call `loadCustomTools` here so factories bind + // to THIS session's `CustomToolAPI` (cwd, exec, pushPendingAction, UI). + // Forwarding the parent's `LoadedCustomTool[]` directly would route tool + // execution back through the parent — wrong for isolated tasks and for + // pending-action queueing. + customToolPaths = + options.preloadedCustomToolPaths ?? + (await logger.time("discoverCustomToolPaths", () => discoverCustomToolPaths([], cwd))); + const customToolsLoadResult = await logger.time("loadCustomTools", () => + loadCustomTools(customToolPaths, cwd, builtInToolNames, action => queueResolveHandler(toolSession, action)), + ); + for (const { path, error } of customToolsLoadResult.errors) { + logger.error("Custom tool load failed", { path, error }); + } + if (customToolsLoadResult.tools.length > 0) { + customTools.push(...customToolsLoadResult.tools.map(loaded => loaded.tool)); + } + + inlineExtensions.push(...(options.extensions ?? [])); + inlineExtensions.push(createAutoresearchExtension); + if (customTools.length > 0) { + inlineExtensions.push(createCustomToolsExtension(customTools)); + } } // Forward the path list (NOT the loaded tools) to subagents so they // re-bind under their own `CustomToolAPI` while skipping the FS scan. toolSession.customToolPaths = customToolPaths; - const inlineExtensions: ExtensionFactory[] = options.extensions ? [...options.extensions] : []; - inlineExtensions.push((await import("./autoresearch")).createAutoresearchExtension); - if (customTools.length > 0) { - inlineExtensions.push(createCustomToolsExtension(customTools)); - } - // Load extensions. Three paths: // 1. `preloadedExtensions` (CLI): caller already loaded — reuse the // Extension instances. Shallow-clone `extensions` so the inline @@ -1832,7 +1961,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // the flag and pre-resolved the result already reflects that choice. let extensionPaths: string[]; let extensionsResult: LoadExtensionsResult; - if (options.preloadedExtensions) { + if (restrictToolNames) { + // Allocate a session runtime without evaluating caller-provided extension + // instances, paths, or factories. + extensionPaths = []; + extensionsResult = await loadExtensions([], cwd, eventBus); + } else if (options.preloadedExtensions) { extensionsResult = { ...options.preloadedExtensions, extensions: [...options.preloadedExtensions.extensions], @@ -1897,8 +2031,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // the background. await modelRegistry.refreshRuntimeProviders("offline"); // Continue runtime discovery in the background (cache-aware) so startup is - // only blocked on local cache reads, not provider network fetches. - void modelRegistry.refreshRuntimeProviders().catch(error => { + // only blocked on local cache reads, not provider network fetches. Stash + // the promise so the deferred `--model` retry below can await it instead + // of starting a second concurrent discovery pass (the unfiltered + // `refresh()` also covers runtime model managers). + const runtimeDiscoveryPromise = modelRegistry.refreshRuntimeProviders().catch(error => { logger.warn("runtime provider discovery failed", { error: error instanceof Error ? error.message : String(error), }); @@ -1944,20 +2081,109 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } // Resolve deferred --model/subagent patterns now that extension models are - // registered. Expand role aliases (`@smol`) and comma chains to concrete - // selectors first so deferred resolution accepts everything the immediate - // path (resolveModelOverride → resolveModelRoleValue) accepts. + // registered. Use the same CLI resolver as the immediate path so bare role + // names, exact model names, and provider selectors keep one precedence rule. if (!model && deferredModelPatterns.length > 0) { - const expandedModelPatterns = resolveConfiguredModelPatterns(deferredModelPatterns, settings); + // Deferred `--model` patterns almost always failed at the immediate + // path (main.ts:881) precisely because discovery-backed providers + // hadn't populated yet. Await the in-flight runtime discovery + // already kicked off above (stash + reuse avoids a second concurrent + // `#refreshRuntimeDiscoveries` pass for the same runtime model + // managers; it resolves instantly when no runtime managers are + // registered). `refreshRuntimeProviders()` only covers runtime model + // managers, not config-discovery providers (e.g. user-configured + // ollama); fall back to a full cache-aware refresh only when the + // runtime pass didn't surface a match AND config-discovery providers + // exist to fetch from. By then runtime managers short-circuit on the + // fresh cache written by the awaited pass, closing the double-fetch + // window. + await logger.time("resolveModelDiscoveryDeferredRetry", () => runtimeDiscoveryPromise); + const availableModelsAfterRuntime = modelRegistry.getAll(); + const runtimeResolved = deferredModelPatterns.some(pattern => + availableModelsAfterRuntime.some(m => `${m.provider}/${m.id}` === pattern), + ); + if (!runtimeResolved && modelRegistry.getDiscoverableProviders().length > 0) { + await logger.time("resolveModelDiscoveryFallbackNonRuntime", () => + modelRegistry.refresh("online-if-uncached"), + ); + } const availableModels = modelRegistry.getAll(); const matchPreferences = getModelMatchPreferences(settings); + const expandedModelPatterns = deferredModelPatterns.flatMap(pattern => + pattern.split(",").flatMap(selector => { + const trimmedSelector = selector.trim(); + if (!trimmedSelector) return []; + const resolved = resolveCliModel({ + cliModel: trimmedSelector, + modelRegistry, + settings, + preferences: matchPreferences, + }); + if (resolved.configuredPatterns && resolved.configuredPatterns.length > 0) { + const primaryPatterns: Array<{ + pattern: string; + retryFallback: InitialRetryFallbackState | undefined; + }> = resolved.configuredPatterns.map(pattern => ({ + pattern, + retryFallback: undefined, + })); + if (!resolved.configuredRole || !settings.get("retry.modelFallback")) { + return primaryPatterns; + } + const fallbackChains = settings.get("retry.fallbackChains"); + const roleFallbacks = + fallbackChains[resolved.configuredRole] ?? + (resolved.configuredRole === "default" ? undefined : fallbackChains.default); + if (!Array.isArray(roleFallbacks)) return primaryPatterns; + const originalSelector = resolved.configuredPatterns[0]; + const parsedOriginal = parseModelString(originalSelector, { + allowMaxSuffix: true, + allowAutoAlias: true, + isLiteralModelId: (provider, id) => modelRegistry.find(provider, id) !== undefined, + }); + const retryFallback: InitialRetryFallbackState = { + role: resolved.configuredRole, + originalSelector, + originalThinkingLevel: parsedOriginal?.thinkingLevel, + }; + return [ + ...primaryPatterns, + ...roleFallbacks + .filter((pattern): pattern is string => typeof pattern === "string") + .map(pattern => ({ pattern, retryFallback })), + ]; + } + if (resolved.model) { + return [ + { + pattern: formatModelSelectorValue( + resolved.selector ?? formatModelStringWithRouting(resolved.model), + resolved.thinkingLevel, + ), + retryFallback: undefined, + }, + ]; + } + return resolveConfiguredModelPatterns([trimmedSelector], settings).map(pattern => ({ + pattern, + retryFallback: undefined, + })); + }), + ); for (let patternIndex = 0; patternIndex < expandedModelPatterns.length; patternIndex += 1) { - const pattern = expandedModelPatterns[patternIndex]; + const { pattern, retryFallback } = expandedModelPatterns[patternIndex]; const primary = parseModelPattern(pattern, availableModels, matchPreferences); - if (!primary.model) continue; + if (!primary.model || (retryFallback && !hasModelAuth(primary.model))) continue; let selectedModel = primary.model; let selectedThinkingLevel = primary.thinkingLevel; let selectedExplicitThinkingLevel = primary.explicitThinkingLevel; + // A chain entry without its own `:level` suffix inherits the + // unavailable primary's configured thinking level, matching + // runtime fallback-chain semantics. + if (retryFallback && !selectedExplicitThinkingLevel && retryFallback.originalThinkingLevel !== undefined) { + selectedThinkingLevel = retryFallback.originalThinkingLevel; + selectedExplicitThinkingLevel = true; + } let authFallbackUsed = false; if (options.modelPatternAuthFallback) { const primaryKey = await modelRegistry.getApiKey(primary.model); @@ -1985,8 +2211,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} ); const seenSelectors = new Set([primarySelector]); const fallbackSelectors: string[] = []; - for (const fallbackPattern of expandedModelPatterns.slice(patternIndex + 1)) { - const fallback = parseModelPattern(fallbackPattern, availableModels, matchPreferences); + for (const fallbackEntry of expandedModelPatterns.slice(patternIndex + 1)) { + const fallback = parseModelPattern(fallbackEntry.pattern, availableModels, matchPreferences); if (!fallback.model) continue; const fallbackSelector = formatModelSelectorValue( formatModelStringWithRouting(fallback.model), @@ -1996,6 +2222,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} seenSelectors.add(fallbackSelector); fallbackSelectors.push(fallbackSelector); } + if (fallbackSelectors.length === 0) { + for (const selector of options.modelPatternDefaultFallbackChain ?? []) { + if (typeof selector !== "string" || seenSelectors.has(selector)) continue; + seenSelectors.add(selector); + fallbackSelectors.push(selector); + } + } if (fallbackSelectors.length > 0) { const modelRoles: Record = {}; const existingRoles = settings.getModelRoles(); @@ -2020,6 +2253,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } model = selectedModel; + initialRetryFallback = retryFallback; modelFallbackMessage = undefined; if (selectedExplicitThinkingLevel) { restoredSessionThinkingLevel = selectedThinkingLevel; @@ -2048,48 +2282,84 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // path-scoped `enabledModels` allow-list when configured. Skip when the // user explicitly requested a model via --model that wasn't found. if (!model && deferredModelPatterns.length === 0) { - // Re-resolve the allowed set: extension factories above may have - // registered providers/models that weren't visible at startup. - const fallbackCandidates = await resolveAllowedModels(modelRegistry, settings, modelMatchPreferences); - - // Retry the default-role lookup against the post-extension allowed - // set. Extension factories register providers AFTER the early - // `defaultRoleSpec` resolution, so a role pointing at an extension - // model (e.g. an openai-compat plugin's `posthog/claude-opus-4-8`) - // returned `undefined` there. Without this retry the next step's - // `pickDefaultAvailableModel` happily replaces the user's configured - // default with a bundled provider's default whenever a stray - // `OPENAI_API_KEY`/`ANTHROPIC_API_KEY` is in the environment. - // (issue #3569) - if (!hasExplicitModel && !defaultRoleSpec.model) { + // Retry the configured default role against the current catalog, + // setting `model` (+ thinking level) when it resolves. Extension + // factories register providers AFTER the early `defaultRoleSpec` + // resolution, and configured discovery providers may still be + // mid-discovery, so a role pointing at such a model (an openai-compat + // plugin's `posthog/claude-opus-4-8`, a models.yml `openai-models-list` + // endpoint) returned `undefined` there. Without this retry the + // `pickDefaultAvailableModel` fallback below happily replaces the + // user's configured default with a bundled provider's default whenever + // a stray `OPENAI_API_KEY`/`ANTHROPIC_API_KEY` is in the environment. + // (issues #3569, #6162) + const tryResolveDefaultRole = async (): Promise => { + if (hasExplicitModel) return false; + // Re-resolve the allowed set: extension factories and discovery + // refreshes above may have registered models not visible earlier. + const fallbackCandidates = await resolveAllowedModels(modelRegistry, settings, modelMatchPreferences); const reResolvedRoleSpec = resolveModelRoleValue(settings.getModelRole("default"), fallbackCandidates, { settings, matchPreferences: modelMatchPreferences, }); - if (reResolvedRoleSpec.model) { - defaultRoleSpec = reResolvedRoleSpec; - const resolvedDefaultModel = reResolvedRoleSpec.model; - model = resolvedDefaultModel; - modelFallbackMessage = undefined; - // Recompute the thinking level against the now-real model. - // `pickInitialThinkingLevel` closes over `defaultRoleSpec`, - // so the role's explicit selector (e.g. `:max`) now applies. - thinkingLevel = pickInitialThinkingLevel(resolvedDefaultModel); - autoThinking = thinkingLevel === AUTO_THINKING; - effectiveThinkingLevel = concreteThinkingLevel(thinkingLevel); - effectiveThinkingLevel = logger.time("resolveThinkingLevelForModel", () => - autoThinking - ? resolveProvisionalAutoLevel(resolvedDefaultModel) - : resolveThinkingLevelForModel(resolvedDefaultModel, effectiveThinkingLevel), - ); - preconnectModelHost(resolvedDefaultModel.baseUrl); - } - } + if (!reResolvedRoleSpec.model) return false; + defaultRoleSpec = reResolvedRoleSpec; + const resolvedDefaultModel = reResolvedRoleSpec.model; + model = resolvedDefaultModel; + modelFallbackMessage = undefined; + // Recompute the thinking level against the now-real model. + // `pickInitialThinkingLevel` closes over `defaultRoleSpec`, + // so the role's explicit selector (e.g. `:max`) now applies. + thinkingLevel = pickInitialThinkingLevel(resolvedDefaultModel); + autoThinking = thinkingLevel === AUTO_THINKING; + effectiveThinkingLevel = concreteThinkingLevel(thinkingLevel); + effectiveThinkingLevel = logger.time("resolveThinkingLevelForModel", () => + autoThinking + ? resolveProvisionalAutoLevel(resolvedDefaultModel) + : resolveThinkingLevelForModel(resolvedDefaultModel, effectiveThinkingLevel), + ); + preconnectModelHost(resolvedDefaultModel.baseUrl); + return true; + }; + + await tryResolveDefaultRole(); if (!model) { - const defaultModel = pickDefaultAvailableModel(fallbackCandidates.filter(hasModelAuth)); - if (defaultModel) { - model = defaultModel; + const fallbackCandidates = await resolveAllowedModels(modelRegistry, settings, modelMatchPreferences); + let pick = pickDefaultAvailableModel(fallbackCandidates.filter(hasModelAuth)); + + // Cold-cache discovery race (issues #6114, #6162): a discovery + // provider (models.yml `openai-models-list`, LM Studio/Ollama/ + // llama.cpp, or an openai-compat proxy) ships no static models, so + // the static+cached catalog resolved nothing above. Background + // discovery in main.ts fires only AFTER createAgentSession returns, + // so on a cache-cold boot the configured default stays unresolved + // and `pick` silently degrades to an unrelated authed provider's + // default (#6162) or "No models available" (#6114) — even though + // `omp models` (which awaits discovery) lists the model. Await one + // cache-aware discovery pass and retry when a default role is + // configured (must win over `pick`) or nothing resolved at all. + // The common path — role already resolved, or a `pick` with no + // configured default — never pays for it. + const defaultRoleConfigured = Boolean(settings.getModelRole("default")); + if ( + !hasExplicitModel && + (defaultRoleConfigured || !pick) && + modelRegistry.getDiscoverableProviders().length > 0 + ) { + await logger.time("resolveModelDiscoveryFallback", () => modelRegistry.refresh("online-if-uncached")); + if (!(await tryResolveDefaultRole()) && !model) { + const refreshedCandidates = await resolveAllowedModels( + modelRegistry, + settings, + modelMatchPreferences, + ); + pick = pickDefaultAvailableModel(refreshedCandidates.filter(hasModelAuth)); + } + } + + if (!model && pick) { + model = pick; } } if (model) { @@ -2141,11 +2411,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } - // Discover custom commands (TypeScript slash commands) - const customCommandsResult: CustomCommandsLoadResult = options.disableExtensionDiscovery - ? { commands: [], errors: [] } - : await logger.time("discoverCustomCommands", loadCustomCommandsInternal, { cwd, agentDir }); - if (!options.disableExtensionDiscovery) { + // Restricted sessions do not discover or evaluate custom command modules. + const customCommandsResult: CustomCommandsLoadResult = + options.disableExtensionDiscovery || restrictToolNames + ? { commands: [], errors: [] } + : await logger.time("discoverCustomCommands", loadCustomCommandsInternal, { cwd, agentDir }); + if (!options.disableExtensionDiscovery && !restrictToolNames) { for (const { path, error } of customCommandsResult.errors) { logger.error("Failed to load custom command", { path, error }); } @@ -2190,8 +2461,10 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }); const toolContextStore = new ToolContextStore(getSessionContext); - const registeredTools = extensionRunner.getAllRegisteredTools(); - const sdkCustomTools = options.customTools?.filter(tool => !isLegacyBuiltinToolDefinition(tool)) ?? []; + const registeredTools = restrictToolNames ? [] : extensionRunner.getAllRegisteredTools(); + const sdkCustomTools = restrictToolNames + ? [] + : (options.customTools?.filter(tool => !isLegacyBuiltinToolDefinition(tool)) ?? []); const allCustomTools = [ ...registeredTools, ...sdkCustomTools.map(tool => { @@ -2214,7 +2487,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} toolRegistry.set(tool.name, tool); builtInRegistryToolNames.add(tool.name); } - if (!toolRegistry.has("goal") && settings.get("goal.enabled")) { + if (!restrictToolNames && !toolRegistry.has("goal") && settings.get("goal.enabled")) { const goalTool = await logger.time("createTools:goal:session", HIDDEN_TOOLS.goal, toolSession); if (goalTool) { toolRegistry.set(goalTool.name, wrapToolWithMetaNotice(goalTool)); @@ -2244,31 +2517,50 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} builtInRegistryToolNames.delete("edit"); } - // Staged actions (tool previews, plan approval) resolve through `write` - // to the resolution devices (`xd://resolve`, `xd://reject`, - // `xd://propose`), so `write` must stay in the registry whenever any - // code path can stage one: a deferrable tool, or plan mode installing a - // plan-proposal handler (issue #1428). Dropping it on read-only sessions - // (e.g. plan-mode toolset `read`, `search`, `find`, `web_search`) leaves - // plan mode unable to exit through the intended path. - const hasDeferrableTools = Array.from(toolRegistry.values()).some(tool => tool.deferrable === true); - const hasXdevTools = (toolSession.xdevRegistry?.size ?? 0) > 0; - const planModeAvailable = settings.get("plan.enabled"); - if ((hasDeferrableTools || hasXdevTools || planModeAvailable) && !toolRegistry.has("write")) { - const writeTool = await logger.time("createTools:write:session", BUILTIN_TOOLS.write, toolSession); - if (writeTool) { + let writeRegistration: Promise | undefined; + const ensureWriteRegistered = (): Promise => { + if (toolRegistry.has("write")) return Promise.resolve(builtInRegistryToolNames.has("write")); + writeRegistration ??= (async () => { + const writeTool = await logger.time("createTools:write:session", BUILTIN_TOOLS.write, toolSession); + if (!writeTool || toolRegistry.has("write")) return builtInRegistryToolNames.has("write"); toolRegistry.set( writeTool.name, new ExtensionToolWrapper(wrapToolWithMetaNotice(writeTool), extensionRunner) as Tool, ); builtInRegistryToolNames.add(writeTool.name); - } + return true; + })().finally(() => { + writeRegistration = undefined; + }); + return writeRegistration; + }; + + // Existing staged/device paths need write registered before active-set assembly. + // Deferred MCP also registers it now, but refresh activates it only after a server connects. + const hasDeferrableTools = Array.from(toolRegistry.values()).some(tool => tool.deferrable === true); + const hasXdevTools = (toolSession.xdevRegistry?.size ?? 0) > 0; + const planModeAvailable = settings.get("plan.enabled"); + if (!restrictToolNames && (hasDeferrableTools || hasXdevTools || planModeAvailable || deferMCPDiscoveryForUI)) { + await ensureWriteRegistered(); } let cursorEventEmitter: ((event: AgentEvent) => void) | undefined; + // Built-in xd:// devices (ast_edit, debug, browser, lsp, web_search) are + // mounted in createTools BEFORE this loop wraps registry entries in + // ExtensionToolWrapper, so the registry holds them unwrapped. The normal + // `write xd://` path runs approval through the wrapped `write` tool's + // tier gate, but Cursor invokes advertised devices via `tool.execute()` + // directly — so wrap unwrapped devices here to keep the approval/deny/prompt + // gate. Dynamic mounts (custom/MCP) already come from the wrapped registry. + const resolveCursorDevice = (name: string): AgentTool | undefined => { + const device = toolSession.xdevRegistry?.get(name); + if (!device) return undefined; + return device instanceof ExtensionToolWrapper ? device : new ExtensionToolWrapper(device, extensionRunner); + }; const cursorExecHandlers = new CursorExecHandlers({ cwd, tools: toolRegistry, + getTool: resolveCursorDevice, getToolContext: () => toolContextStore.getContext(), emitEvent: event => cursorEventEmitter?.(event), }); @@ -2288,8 +2580,10 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} ): Promise => { toolContextStore.setToolNames(toolNames); const promptTools = buildSystemPromptToolMetadata(tools); - const memoryBackend = await resolveMemoryBackend(settings); - const memoryInstructions = await memoryBackend.buildDeveloperInstructions(agentDir, settings, session); + const memoryBackend = restrictToolNames ? undefined : await resolveMemoryBackend(settings); + const memoryInstructions = memoryBackend + ? await memoryBackend.buildDeveloperInstructions(agentDir, settings, session) + : undefined; // Build combined append prompt: memory instructions + auto-learn guidance // + MCP server instructions. For UI sessions MCP discovery is deferred, so @@ -2304,10 +2598,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // session-start build — so a subagent that filtered them out, a mid-session // enable that never built them, or a same-named custom tool while auto-learn // is off all get no guidance. - const autoLearnInstructions = buildAutoLearnInstructions({ - manageSkill: builtInToolNames.includes("manage_skill"), - learn: builtInToolNames.includes("learn"), - }); + const autoLearnInstructions = restrictToolNames + ? undefined + : buildAutoLearnInstructions({ + manageSkill: builtInToolNames.includes("manage_skill"), + learn: builtInToolNames.includes("learn"), + }); const appendParts: string[] = []; if (memoryInstructions) appendParts.push(memoryInstructions); if (autoLearnInstructions) appendParts.push(autoLearnInstructions); @@ -2339,9 +2635,9 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} cwd, xdevTools: toolSession.xdevRegistry?.entries() ?? [], xdevDocs: toolSession.xdevRegistry?.docsAll() ?? "", - autoQaEnabled: isAutoQaEnabled(settings), + autoQaEnabled: !restrictToolNames && isAutoQaEnabled(settings), resolvedCustomPrompt: options.customSystemPrompt, - skills, + skills: session?.skills ?? skills, contextFiles, tools: promptTools, toolNames, @@ -2356,11 +2652,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} eagerTasksAlways, taskBatch: settings.get("task.batch"), taskMaxConcurrency: settings.get("task.maxConcurrency"), - taskIrcEnabled: isIrcEnabled(settings, options.taskDepth ?? 0), + taskIrcEnabled: !restrictToolNames && isIrcEnabled(settings, options.taskDepth ?? 0), secretsEnabled, workspaceTree: workspaceTreePromise, includeWorkspaceTree, - memoryRootEnabled: memoryBackend.id === "local", + memoryRootEnabled: memoryBackend?.id === "local", model: getActiveModelString(), includeModelInPrompt: settings.get("includeModelInPrompt"), personality: agentKind === "sub" ? "none" : settings.get("personality"), @@ -2401,7 +2697,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // exactly the builtins createTools built (`builtInToolNames` — provenance, so a // same-named custom/extension tool is never force-activated when auto-learn is // off) to keep guidance, controller, and the active set consistent. - if (explicitlyRequestedToolNames) { + if (!restrictToolNames && explicitlyRequestedToolNames) { for (const name of ["manage_skill", "learn"]) { if (builtInToolNames.includes(name) && !explicitlyRequestedToolNames.includes(name)) { explicitlyRequestedToolNames.push(name); @@ -2410,21 +2706,29 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } const requestedToolNames = explicitlyRequestedToolNames ?? toolNamesFromRegistry; const normalizedRequested = requestedToolNames.filter(name => toolRegistry.has(name)); - const requestedToolNameSet = new Set(normalizedRequested); const defaultInactiveToolNames = new Set( registeredTools.filter(tool => tool.definition.defaultInactive).map(tool => tool.definition.name), ); const requestedActiveToolNames = normalizedRequested.filter(name => name !== "goal"); + const explicitlyRequestedToolNameSet = explicitlyRequestedToolNames + ? new Set(explicitlyRequestedToolNames) + : undefined; + const xdevReadAvailable = + builtInRegistryToolNames.has("read") && + (explicitlyRequestedToolNameSet === undefined || explicitlyRequestedToolNameSet.has("read")); const initialRequestedActiveToolNames = options.toolNames ? requestedActiveToolNames : requestedActiveToolNames.filter(name => !defaultInactiveToolNames.has(name)); let initialToolNames = [...initialRequestedActiveToolNames]; - // Custom tools and extension-registered tools are always included regardless of toolNames filter - const alwaysInclude: string[] = [ - ...sdkCustomTools.map(t => (isCustomTool(t) ? t.name : t.name)), - ...registeredTools.filter(t => !t.definition.defaultInactive).map(t => t.definition.name), - ]; + // Custom tools and extension-registered tools are always included regardless of toolNames filter. + // Restricted callers own the list, so never widen it with registered tools. + const alwaysInclude: string[] = restrictToolNames + ? [] + : [ + ...sdkCustomTools.map(t => (isCustomTool(t) ? t.name : t.name)), + ...registeredTools.filter(t => !t.definition.defaultInactive).map(t => t.definition.name), + ]; for (const name of alwaysInclude) { if (toolRegistry.has(name) && !initialToolNames.includes(name)) { initialToolNames.push(name); @@ -2446,24 +2750,32 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }); hasRegistered = true; - // Partition the initial enabled set for the xd:// transport: discoverable - // tools become mounted devices; the rest stay top-level. The registry - // already holds the built-in devices (mounted in createTools); this - // reconciles the initial dynamic mounts (image-gen, TTS, startup MCP, - // active extension tools) and drops them from the top-level names so they - // never ship a schema. Presentation only — selection already happened. + // Partition the initial enabled set for the xd:// transport: ambient + // discoverable tools become mounted devices, while explicitly requested + // tools keep their top-level presentation. The registry already holds the + // default-set built-in devices from createTools; this reconciles dynamic + // mounts (image-gen, TTS, startup MCP, active extension tools). let initialMountedXdevToolNames: string[] = []; if (toolSession.xdevRegistry) { const topLevelToolNames: string[] = []; const mountedTools: Tool[] = []; for (const name of initialToolNames) { const tool = toolRegistry.get(name); - if (tool && isMountableUnderXdev(tool)) mountedTools.push(tool); + const explicitlyRequested = explicitlyRequestedToolNameSet?.has(name) === true; + if (tool && xdevReadAvailable && !explicitlyRequested && isMountableUnderXdev(tool)) + mountedTools.push(tool); else topLevelToolNames.push(name); } - toolSession.xdevRegistry.reconcile(mountedTools); - initialMountedXdevToolNames = mountedTools.map(tool => tool.name); - initialToolNames = topLevelToolNames; + const writeTransportAvailable = mountedTools.length === 0 || (await ensureWriteRegistered()); + if (writeTransportAvailable) { + toolSession.xdevRegistry.reconcile(mountedTools); + initialMountedXdevToolNames = mountedTools.map(tool => tool.name); + initialToolNames = topLevelToolNames; + if (initialMountedXdevToolNames.length > 0 && !initialToolNames.includes("write")) + initialToolNames.push("write"); + } else { + toolSession.xdevRegistry.reconcile([]); + } } setActiveToolNames(initialToolNames); @@ -2534,8 +2846,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} if (snapcompactInline) transformed = await snapcompactInline.transform(transformed, transformModel); return clampProviderContextImages(transformed, transformModel); }; - const onPayload = async (payload: unknown, _model?: Model) => { - return await extensionRunner.emitBeforeProviderRequest(payload); + const onPayload = async (payload: unknown, model?: Model) => { + return await extensionRunner.emitBeforeProviderRequest(payload, model); }; const onResponse: SimpleStreamOptions["onResponse"] = async (response, model) => { await extensionRunner.emitAfterProviderResponse(response, model); @@ -2548,6 +2860,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const initialTools = initialToolNames .map(name => toolRegistry.get(name)) .filter((tool): tool is AgentTool => tool !== undefined); + const autoLearnCaptureTools = initialTools.filter(tool => tool.name === "manage_skill" || tool.name === "learn"); const openaiWebsocketSetting = settings.get("providers.openaiWebsockets") ?? "off"; const preferOpenAICodexWebsockets = @@ -2574,6 +2887,19 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} settings, createSettingsAwareStreamFn(settings), ); + const transformToolCallArguments = (args: Record): Record => { + let result = args; + const maxTimeout = settings.get("tools.maxTimeout"); + if (maxTimeout > 0 && typeof result.timeout === "number") { + result = { ...result, timeout: Math.min(result.timeout, maxTimeout) }; + } + if (obfuscator?.hasSecrets()) { + result = deobfuscateToolArguments(obfuscator, result); + } + return result; + }; + const kimiApiFormatSetting = settings.get("providers.kimiApiFormat"); + const kimiApiFormat = kimiApiFormatSetting === "auto" ? undefined : kimiApiFormatSetting; agent = new Agent({ initialState: { systemPrompt, @@ -2607,7 +2933,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} presencePenalty: settings.get("presencePenalty") >= 0 ? settings.get("presencePenalty") : undefined, repetitionPenalty: settings.get("repetitionPenalty") >= 0 ? settings.get("repetitionPenalty") : undefined, hideThinkingSummary: settings.get("omitThinking"), - kimiApiFormat: settings.get("providers.kimiApiFormat") ?? "anthropic", + kimiApiFormat, preferWebsockets: preferOpenAICodexWebsockets, getToolContext: tc => toolContextStore.getContext(tc), getApiKey: requestModel => modelRegistry.resolver(requestModel, agent.sessionId), @@ -2626,17 +2952,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} return settingsAwareStreamFn(streamModel, context, streamOptions); }, cursorExecHandlers, - transformToolCallArguments: (args, _toolName) => { - let result = args; - const maxTimeout = settings.get("tools.maxTimeout"); - if (maxTimeout > 0 && typeof result.timeout === "number") { - result = { ...result, timeout: Math.min(result.timeout, maxTimeout) }; - } - if (obfuscator?.hasSecrets()) { - result = deobfuscateToolArguments(obfuscator, result); - } - return result; - }, + getCursorTools: () => [...(toolSession.xdevRegistry?.list() ?? [])], + transformToolCallArguments, intentTracing: !!intentField, pruneToolDescriptions: inlineToolDescriptors, dialect: resolveDialect(settings.get("tools.format"), model), @@ -2717,6 +3034,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} agent, pruneToolDescriptions: inlineToolDescriptors, thinkingLevel: autoThinking ? AUTO_THINKING : effectiveThinkingLevel, + initialRetryFallback, prewalk: options.prewalk, planYolo: options.planYolo, serviceTierByFamily: initialServiceTierByFamily, @@ -2737,6 +3055,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} customCommands: customCommandsResult.commands, skills, skillWarnings, + skillsReloadable: options.skills === undefined, skillsSettings: settings.getGroup("skills"), modelRegistry, toolRegistry, @@ -2757,8 +3076,9 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} getXdevToolEntries: () => toolSession.xdevRegistry?.entries() ?? [], xdevRegistry: toolSession.xdevRegistry, initialMountedXdevToolNames, - requestedToolNames: requestedToolNameSet, + presentationPinnedToolNames: explicitlyRequestedToolNameSet, setActiveToolNames, + ensureWriteRegistered, getMcpServerInstructions: mcpManager ? () => { const raw = mcpManager.getServerInstructions(); @@ -2918,8 +3238,56 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }); }; - // Auto-learn can immediately trigger a synthetic capture turn after the - // first real stop. When a memory backend is selected, install that backend's + const runAutoLearnCapture = createAutoLearnCaptureRunner({ + sourceAgent: agent, + captureTools: autoLearnCaptureTools, + onPayload, + onResponse, + createAgent: captureOptions => { + const captureModel = captureOptions.initialState?.model; + const captureSessionId = captureOptions.sessionId; + if (!captureModel || !captureSessionId) throw new Error("Auto-learn capture identity is incomplete"); + return new Agent({ + ...captureOptions, + cwd: sessionManager.getCwd(), + cwdResolver: () => sessionManager.getCwd(), + convertToLlm: convertToLlmFinal, + transformContext: async messages => wrapSteeringForModel(messages), + transformProviderContext: async (context, transformModel) => { + const transformed = obfuscator ? obfuscateProviderContext(obfuscator, context) : context; + return clampProviderContextImages(transformed, transformModel); + }, + thinkingBudgets: agent.thinkingBudgets, + temperature: agent.temperature, + topP: agent.topP, + topK: agent.topK, + minP: agent.minP, + presencePenalty: agent.presencePenalty, + repetitionPenalty: agent.repetitionPenalty, + serviceTierResolver: agent.serviceTierResolver, + hideThinkingSummary: agent.hideThinkingSummary, + maxRetryDelayMs: agent.maxRetryDelayMs, + kimiApiFormat, + preferWebsockets: preferOpenAICodexWebsockets, + getToolContext: toolCall => toolContextStore.getContext(toolCall), + streamFn: settingsAwareStreamFn, + transformToolCallArguments, + intentTracing: !!intentField, + pruneToolDescriptions: inlineToolDescriptors, + dialect: resolveDialect(settings.get("tools.format"), captureModel), + abortOnFabricatedToolResult: settings.get("tools.abortOnFabricatedResult"), + appendOnlyContext: shouldEnableAppendOnlyContext( + settings.get("provider.appendOnlyContext"), + captureModel, + ) + ? new AppendOnlyContextManager() + : undefined, + }); + }, + }); + + // Auto-learn can immediately trigger a private capture after the first real + // stop. When a memory backend is selected, install that backend's // per-session state first so the capture turn's `learn` tool observes the // same initialized state as normal memory tools. Other sessions keep memory // startup in the background to preserve the existing startup profile. @@ -2932,11 +3300,17 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // and the tools; the fire-time re-check in `#onAgentEnd` still handles a // mid-session DISABLE. The subscription lives for the session's lifetime; the // reference is intentionally discarded (the listener retains it). - if (settings.get("autolearn.enabled") && taskDepth === 0) { - await logger.time("startMemoryStartupTask", startMemoryBackend); - new AutoLearnController({ session, settings }); - } else { - void logger.time("startMemoryStartupTask", startMemoryBackend); + if (!restrictToolNames) { + if (settings.get("autolearn.enabled") && taskDepth === 0) { + await logger.time("startMemoryStartupTask", startMemoryBackend); + new AutoLearnController({ + session, + settings, + capture: content => session.runAutolearnCapture(signal => runAutoLearnCapture(content, signal)), + }); + } else { + void logger.time("startMemoryStartupTask", startMemoryBackend); + } } // Wire MCP manager callbacks to session for reactive tool updates. diff --git a/packages/coding-agent/src/session/agent-session-error-log.test.ts b/packages/coding-agent/src/session/agent-session-error-log.test.ts new file mode 100644 index 000000000..b68871827 --- /dev/null +++ b/packages/coding-agent/src/session/agent-session-error-log.test.ts @@ -0,0 +1,59 @@ +/** + * Contract: a turn ending in a provider error surfaces one warn-level log + * carrying `provider`/`model`/`errorMessage`/`errorStatus`/`errorId`, so + * recurring provider stream failures are diagnosable from the main log alone + * (issue #6177). Non-error stops must not emit it. + */ +import { afterAll, afterEach, describe, expect, it } from "bun:test"; +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; +import { logger } from "@oh-my-pi/pi-utils"; +import { logProviderTurnError } from "./agent-session"; + +function makeMessage(overrides: Partial): AssistantMessage { + return { + role: "assistant", + content: [], + api: "cursor", + provider: "cursor", + model: "composer-2.5", + usage: {} as AssistantMessage["usage"], + stopReason: "error", + timestamp: 1, + ...overrides, + }; +} + +describe("logProviderTurnError", () => { + const events: logger.LogEvent[] = []; + const dispose = logger.registerLogSink(event => events.push(event)); + + afterEach(() => { + events.length = 0; + }); + afterAll(() => dispose()); + + it("emits a warn log with the provider error fields for stopReason:error", () => { + logProviderTurnError( + makeMessage({ + errorMessage: "stream stall: idle watchdog fired", + errorStatus: 500, + errorId: 42, + }), + ); + + const warn = events.find(e => e.level === "warn" && e.message === "agent turn ended with provider error"); + expect(warn).toBeDefined(); + expect(warn?.context).toMatchObject({ + provider: "cursor", + model: "composer-2.5", + errorMessage: "stream stall: idle watchdog fired", + errorStatus: 500, + errorId: 42, + }); + }); + + it("does not emit for a successful stop", () => { + logProviderTurnError(makeMessage({ stopReason: "stop", errorMessage: undefined })); + expect(events.some(e => e.message === "agent turn ended with provider error")).toBe(false); + }); +}); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 9ce7adc89..1a9dc9b14 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -30,6 +30,8 @@ import { type AgentMessage, type AgentState, type AgentTool, + type AgentToolCall, + type AgentToolContext, type AgentToolResult, type AgentTurnEndContext, AppendOnlyContextManager, @@ -37,6 +39,7 @@ import { type CompactionSummaryMessage, countTokens, createToolScopedAbortReason, + isSyntheticToolResultMessage, resolveTelemetry, type StreamFn, TERMINAL_TOOL_RESULT_ABORT_REASON, @@ -63,6 +66,7 @@ import { estimateTokens, generateBranchSummary, generateHandoffFromContext, + invalidateMessageCache, prepareCompaction, renderHandoffPrompt, resolveBudgetReserveTokens, @@ -104,6 +108,7 @@ import type { TextContent, ToolCall, ToolChoice, + ToolResultMessage, Usage, UsageReport, } from "@oh-my-pi/pi-ai"; @@ -121,6 +126,7 @@ import { } from "@oh-my-pi/pi-ai"; import * as AIError from "@oh-my-pi/pi-ai/error"; import { resetOpenAICodexHistoryAfterCompaction } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import { kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; import { toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema"; import { GeminiHeaderRunDetector, isGeminiThinkingModel } from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { type RepeatedToolCallDetection, ToolCallLoopGuard } from "@oh-my-pi/pi-ai/utils/tool-call-loop-guard"; @@ -137,6 +143,7 @@ import { getInstallId, isBunTestRuntime, isEnoent, + isInteractiveHost, logger, postmortem, prompt, @@ -155,6 +162,7 @@ import { type AdvisorNote, AdvisorOutputQuarantinedError, AdvisorRuntime, + type AdvisorRuntimeStatus, type AdvisorSeverity, AdvisorTranscriptRecorder, advisorTranscriptFilename, @@ -170,6 +178,7 @@ import { } from "../advisor"; import { type AsyncJob, type AsyncJobDeliveryState, AsyncJobManager } from "../async"; import { classifyDifficulty } from "../auto-thinking/classifier"; +import { reset as resetCapabilities } from "../capability"; import type { Rule } from "../capability/rule"; import { shouldEnableAppendOnlyContext } from "../config/append-only-context-mode"; import type { ModelRegistry } from "../config/model-registry"; @@ -196,6 +205,7 @@ import { onModelRolesChanged, validateProviderMaxInFlightRequests, } from "../config/settings"; +import { CursorExecHandlers } from "../cursor"; import { RawSseDebugBuffer } from "../debug/raw-sse-buffer"; import { expandApplyPatchToEntries, normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit"; import { getFileSnapshotStore } from "../edit/file-snapshot-store"; @@ -232,12 +242,14 @@ import type { TurnEndEvent, TurnStartEvent, } from "../extensibility/extensions"; +import { emitSessionShutdownEvent } from "../extensibility/extensions"; +import { ManagedTimers } from "../extensibility/extensions/managed-timers"; import { createExtensionModelQuery } from "../extensibility/extensions/model-api"; import type { CompactOptions, ContextUsage } from "../extensibility/extensions/types"; import { ExtensionToolWrapper } from "../extensibility/extensions/wrapper"; import type { HookCommandContext } from "../extensibility/hooks/types"; import type { RecoveredRetryError } from "../extensibility/shared-events"; -import type { Skill, SkillWarning } from "../extensibility/skills"; +import { loadSkills, type Skill, type SkillWarning, setActiveSkills } from "../extensibility/skills"; import { expandSlashCommand, type FileSlashCommand } from "../extensibility/slash-commands"; import { GoalRuntime } from "../goals/runtime"; import type { Goal, GoalModeState } from "../goals/state"; @@ -251,9 +263,13 @@ import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate"; import { theme } from "../modes/theme/theme"; import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; -import { computeNonMessageBreakdown, computeNonMessageTokens } from "../modes/utils/context-usage"; +import { + computeNonMessageBreakdown, + computeNonMessageTokens, + estimateToolSchemaTokens, +} from "../modes/utils/context-usage"; import { containsWorkflow, renderWorkflowNotice } from "../modes/workflow"; -import { resolveApprovedPlan } from "../plan-mode/approved-plan"; +import { type PlanApprovalDetails, resolveApprovedPlan } from "../plan-mode/approved-plan"; import { createPlanReadMatcher } from "../plan-mode/plan-protection"; import type { PlanModeState } from "../plan-mode/state"; import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" }; @@ -291,6 +307,7 @@ import { AgentRegistry } from "../registry/agent-registry"; import { deobfuscateAssistantContent, deobfuscateSessionContext, + deobfuscateToolArguments, obfuscateMessages, obfuscateProviderContext, type SecretObfuscator, @@ -309,12 +326,13 @@ import { } from "../thinking"; import { formatTitleConversationContext, type TitleConversationTurn } from "../tiny/message-preproc"; import { shutdownTinyTitleClient } from "../tiny/title-client"; +import { type AskToolDetails, type AskToolInput, recoverAskQuestions } from "../tools/ask"; import { assertEditableFile } from "../tools/auto-generated-guard"; import { releaseTabsForOwner } from "../tools/browser/tab-supervisor"; import { isMCPToolName, normalizeToolNames } from "../tools/builtin-names"; import type { CheckpointState, CompletedRewindState } from "../tools/checkpoint"; import { outputMeta, wrapToolWithMetaNotice } from "../tools/output-meta"; -import { normalizeLocalScheme, resolveToCwd } from "../tools/path-utils"; +import { isInternalUrlPath, normalizeLocalScheme, resolveToCwd } from "../tools/path-utils"; import { buildResolveReminderMessage, isPreviewResolutionToolCall, @@ -392,6 +410,34 @@ import { YieldQueue } from "./yield-queue"; const SESSION_STOP_CONTINUATION_CAP = 8; const PLAN_MODE_REMINDER_MAX = 3; +type BashAppendDestination = + | { kind: "current"; manager: SessionManager } + | { kind: "detached"; manager: SessionManager } + | { kind: "branch"; manager: SessionManager; parentId: string | null }; + +interface BashSessionTarget { + sessionId: string; + refs: number; + destination?: BashAppendDestination; + pending?: Promise; +} + +interface PendingBashMessage { + target: BashSessionTarget; + message: BashExecutionMessage; +} + +interface BashSessionTransition { + oldTarget: BashSessionTarget; + newTarget: BashSessionTarget; + oldSessionId: string; + oldSessionFile: string | undefined; + oldLeafId: string | null; + detachedManager: SessionManager | undefined; + resolveOld: ((destination: BashAppendDestination) => void) | undefined; + resolveNew: (destination: BashAppendDestination) => void; +} + /** * Mutating tool results (`bash`/`eval`/`edit`/`write`/`ast_edit`) without the * agent touching the `todo` tool that trip the mid-run reconciliation nudge. @@ -530,6 +576,43 @@ function reportFromRewindReportContent(content: string): string { return report.trim(); } +type SemanticCheckpointToolName = "checkpoint" | "rewind"; + +type SemanticToolResult = { + toolName: SemanticCheckpointToolName; + details?: unknown; +}; + +/** + * Normalize checkpoint/rewind results across native calls and `write xd://` + * dispatches. Xdev keeps the wrapped tool's result details under `xdev.inner`, + * while direct calls put them on the result itself. + */ +function semanticToolResult(toolName: string | undefined, result: unknown): SemanticToolResult | undefined { + if (toolName === "checkpoint" || toolName === "rewind") { + const details = result && typeof result === "object" && "details" in result ? result.details : undefined; + return { toolName, details }; + } + const dispatch = writeDeviceDispatch(toolName ?? "", result); + if (dispatch?.mode !== "execute" || (dispatch.tool !== "checkpoint" && dispatch.tool !== "rewind")) { + return undefined; + } + return { toolName: dispatch.tool, details: dispatch.inner }; +} + +function isTodoPhase(value: unknown): value is TodoPhase { + if (!isRecord(value) || typeof value.name !== "string" || !Array.isArray(value.tasks)) return false; + return value.tasks.every( + task => + isRecord(task) && + typeof task.content === "string" && + (task.status === "pending" || + task.status === "in_progress" || + task.status === "completed" || + task.status === "abandoned"), + ); +} + function completedRewindFromEntry(entry: SessionEntry): CompletedRewindState | undefined { if (entry.type !== "custom_message" || entry.customType !== "rewind-report") return undefined; const details = entry.details; @@ -542,21 +625,18 @@ function completedRewindFromEntry(entry: SessionEntry): CompletedRewindState | u reportFromRewindReportContent(customMessageContentText(entry.content)); return report.length > 0 ? { report, startedAt, rewoundAt } : undefined; } - -function isSuccessfulCheckpointEntry(entry: SessionEntry): entry is SessionMessageEntry & { - message: { role: "toolResult"; toolName: "checkpoint"; isError?: false }; -} { - return ( - entry.type === "message" && - entry.message.role === "toolResult" && - entry.message.toolName === "checkpoint" && - entry.message.isError !== true - ); +function isSuccessfulCheckpointEntry( + entry: SessionEntry, +): entry is SessionEntry & { type: "message"; message: Extract } { + if (entry.type !== "message" || entry.message.role !== "toolResult" || entry.message.isError === true) { + return false; + } + return semanticToolResult(entry.message.toolName, entry.message)?.toolName === "checkpoint"; } function checkpointStartedAtFromEntry(entry: SessionEntry): string | undefined { if (!isSuccessfulCheckpointEntry(entry)) return undefined; - const details = entry.message.details; + const details = semanticToolResult(entry.message.toolName, entry.message)?.details; if (details && typeof details === "object") { const startedAt = stringProperty(details, "startedAt"); if (startedAt) return startedAt; @@ -670,7 +750,7 @@ const COMPACTION_CHECK_NONE: CompactionCheckResult = { }; const COMPACTION_CHECK_DEFERRED_HANDOFF: CompactionCheckResult = { deferredHandoff: true, - continuationScheduled: true, + continuationScheduled: false, }; const COMPACTION_CHECK_CONTINUATION: CompactionCheckResult = { deferredHandoff: false, @@ -799,6 +879,16 @@ export interface PlanYolo { // Types // ============================================================================ +/** Identifies a retry fallback chain already entered during startup model resolution. */ +export interface InitialRetryFallbackState { + /** Role whose configured primary was unavailable. */ + role: string; + /** Configured primary selector retained for restoration when it becomes available. */ + originalSelector: string; + /** Thinking selector configured for the unavailable primary. */ + originalThinkingLevel: ConfiguredThinkingLevel | undefined; +} + export interface AgentSessionConfig { agent: Agent; sessionManager: SessionManager; @@ -809,6 +899,8 @@ export interface AgentSessionConfig { scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; /** Initial session thinking selector. */ thinkingLevel?: ConfiguredThinkingLevel; + /** Retry chain ownership when startup selected one of its fallback entries. */ + initialRetryFallback?: InitialRetryFallbackState; /** Prewalk from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ prewalk?: Prewalk; /** Force read-only plan mode at start, auto-approve on the model's first @@ -827,6 +919,8 @@ export interface AgentSessionConfig { skills?: Skill[]; /** Skill loading warnings (already captured by SDK) */ skillWarnings?: SkillWarning[]; + /** Whether runtime reloads may rediscover disk-backed skills for this session. */ + skillsReloadable?: boolean; /** Custom commands (TypeScript slash commands) */ customCommands?: LoadedCustomCommand[]; skillsSettings?: SkillsSettings; @@ -840,6 +934,8 @@ export interface AgentSessionConfig { builtInToolNames?: Iterable; /** Update tool-session predicates that render guidance from the live active tool set. */ setActiveToolNames?: (names: Iterable) => void; + /** Register the write transport lazily when runtime xdev mounts first need it. */ + ensureWriteRegistered?: () => Promise; /** Current session pre-LLM message transform pipeline */ transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise; /** @@ -886,7 +982,8 @@ export interface AgentSessionConfig { xdevRegistry?: XdevRegistry; /** Discoverable tools mounted under `xd://` in the initial enabled set (startup partition in `sdk.ts`). */ initialMountedXdevToolNames?: string[]; - requestedToolNames?: ReadonlySet; + /** Explicit/effective names pinned top-level during runtime repartitioning. */ + presentationPinnedToolNames?: ReadonlySet; /** * Optional accessor for live MCP server instructions. Read by the session's * `rebuildSystemPrompt`-skip optimization to detect server-side instruction @@ -1114,15 +1211,20 @@ export interface AdvisorStats { advisors: PerAdvisorStat[]; } -/** One advisor's slice of {@link AdvisorStats}, surfaced for the multi-advisor status panel. */ +/** One advisor's slice of {@link AdvisorStats}. Active advisors carry full + * token/cost data; disabled/no-model/quota-exhausted advisors appear with + * just `name` + `status` so the status line can render a dot for every + * configured advisor. */ export interface PerAdvisorStat { name: string; - model: Model; + status: AdvisorRuntimeStatus; + model?: Model; contextWindow: number; contextTokens: number; tokens: AdvisorStats["tokens"]; cost: number; messages: AdvisorStats["messages"]; + sessionId?: string; } /** @@ -1131,6 +1233,13 @@ export interface PerAdvisorStat { * primary-scoped state (turn counters, interrupt latches, the shared yield * channel) stays on the session. */ +interface AdvisorRetryFallbackState { + role: string; + originalSelector: string; + originalThinkingLevel: ThinkingLevel; + lastAppliedThinkingLevel: ThinkingLevel; +} + interface ActiveAdvisor { /** Display name from config ("default" for the legacy no-YAML advisor). */ name: string; @@ -1147,10 +1256,23 @@ interface ActiveAdvisor { agentUnsubscribe?: () => void; model: Model; thinkingLevel: ThinkingLevel; + /** Provider credential/session identity retained across advisor model switches. */ + providerSessionId: string | undefined; + /** Active chain state retained until the configured primary can be restored. */ + retryFallback?: AdvisorRetryFallbackState; + /** A switched advisor model has not yet completed its first successful turn. */ + retryFallbackPendingSuccess: boolean; /** Stable key for the resolved runtime inputs that require a rebuild to change. */ signature: string; } +/** Runtime-only advisor compaction metadata. It never enters the model-facing summary text. */ +interface AdvisorCompactionSummaryMessage extends CompactionSummaryMessage { + firstKeptEntryId?: string; + /** First message index eligible to anchor provider usage after this compaction. */ + advisorUsageAnchorStartIndex?: number; +} + /** Resolved advisor config ready to instantiate as an {@link ActiveAdvisor}. */ interface AdvisorRuntimeDescriptor { config: AdvisorConfig; @@ -1226,13 +1348,31 @@ function isRetryFallbackModelKey(key: string): boolean { } /** - * A `provider/*` fallback-chain key: matches any active model of that provider, - * so one entry covers every current and future model behind the provider. + * A wildcard fallback-chain key/entry: `provider/*` matches any model of that + * provider; an id-prefixed `provider/prefix/*` (e.g. `openrouter/google/*`) + * scopes it to ids under that prefix — aggregators namespace model ids by + * upstream vendor. */ function isRetryFallbackWildcardKey(key: string): boolean { return key.endsWith("/*"); } +/** + * Split a `…/*` wildcard key/entry into its provider and optional id prefix + * (`google-vertex/*` → provider only; `openrouter/google/*` → provider + * `openrouter`, prefix `google`). A template that names a known provider in + * full wins over the split, so provider ids containing `/` keep working. + */ +function parseRetryFallbackWildcard( + key: string, + isKnownProvider: (provider: string) => boolean, +): { provider: string; idPrefix: string | undefined } { + const template = key.slice(0, -2); + const slash = template.indexOf("/"); + if (slash < 0 || isKnownProvider(template)) return { provider: template, idPrefix: undefined }; + return { provider: template.slice(0, slash), idPrefix: template.slice(slash + 1) }; +} + function formatRetryFallbackSelector(model: Model, thinkingLevel: ThinkingLevel | undefined): string { return formatModelSelectorValue(formatModelStringWithRouting(model), thinkingLevel); } @@ -1536,15 +1676,20 @@ function queuedTextContent(message: AgentMessage): string | undefined { if (!("content" in message)) return undefined; const content = message.content; if (typeof content === "string") return content; - return content.find((part): part is TextContent => part.type === "text")?.text; + for (const part of content) { + if (part.type === "text") return part.text; + } + return undefined; } function queuedImageContent(message: AgentMessage): ImageContent[] | undefined { if (!("content" in message) || typeof message.content === "string") return undefined; - const images = message.content.filter( - (part): part is ImageContent => - part.type === "image" && typeof part.data === "string" && typeof part.mimeType === "string", - ); + const images: ImageContent[] = []; + for (const part of message.content) { + if (part.type === "image" && typeof part.data === "string" && typeof part.mimeType === "string") { + images.push(part); + } + } return images.length > 0 ? images : undefined; } @@ -1556,6 +1701,25 @@ function isAdvisorCard(message: AgentMessage): message is CustomMessage { return message.role === "custom" && message.customType === "advisor"; } +/** + * Emit a warn-level log for a turn that ended in a provider error so recurring + * stream failures are diagnosable from the main log alone. The `agent_end` + * routing trace is debug-only and omits the error fields; without this a + * session dying on provider errors leaves only `stopReason:"error"` debug lines + * and the real cause lives solely in the session transcript (issue #6177). + * No-op for any non-error stop reason. + */ +export function logProviderTurnError(msg: AssistantMessage): void { + if (msg.stopReason !== "error") return; + logger.warn("agent turn ended with provider error", { + provider: msg.provider, + model: msg.model, + errorMessage: msg.errorMessage, + errorStatus: msg.errorStatus, + errorId: msg.errorId, + }); +} + function isTerminalTextAssistantAnswer(message: AgentMessage | undefined): message is AssistantMessage { if (message?.role !== "assistant" || message.stopReason !== "stop") return false; let hasText = false; @@ -1717,6 +1881,22 @@ function titleConversationTurnFromMessage(message: AgentMessage): TitleConversat return { role: message.role, ...(text ? { text } : {}), ...(thinking ? { thinking } : {}) }; } +function syntheticToolResultTailStart(messages: readonly AgentMessage[]): number { + let index = messages.length; + while (index > 0 && isSyntheticToolResultMessage(messages[index - 1])) { + index--; + } + return index; +} + +function retryableAssistantTurnEnd(messages: readonly AgentMessage[]): number | undefined { + const turnEnd = syntheticToolResultTailStart(messages); + const message = messages[turnEnd - 1]; + if (message?.role !== "assistant") return undefined; + if (message.stopReason !== "error" && message.stopReason !== "aborted") return undefined; + return turnEnd; +} + export class AgentSession { readonly agent: Agent; readonly sessionManager: SessionManager; @@ -1773,6 +1953,8 @@ export class AgentSession { * suppresses advisor concern/blocker auto-resume until the user next resumes. * Advisor advice is still recorded into the transcript, just not auto-run. */ #advisorAutoResumeSuppressed = false; + /** Print-mode sessions preserve advisor notes without starting hidden primary turns. */ + #preserveAdvisorAdvice = false; #advisorPrimaryTurnsCompleted = 0; #advisorInterruptImmuneTurnStart: number | undefined; #planModeState: PlanModeState | undefined; @@ -1789,6 +1971,11 @@ export class AgentSession { #advisors: ActiveAdvisor[] = []; /** Configured advisor roster from WATCHDOG.yml; undefined/empty → single legacy advisor. */ #advisorConfigs?: AdvisorConfig[]; + /** Per-advisor runtime status (slug → {name, status}). Tracks disabled/quota/states + * for the configured roster even when the advisor has no live runtime. The name + * is stored alongside the status so {@link getAdvisorStats} doesn't need to + * recompute slugs or resolve config names. */ + #advisorStatuses: Map = new Map(); /** Provider-facing UUIDv7 identities keyed by primary provider session and advisor slug. */ #advisorProviderSessionIds = new Map(); /** Aggregate of the most recent stop's recorder closes; awaited by dispose() and @@ -1852,11 +2039,13 @@ export class AgentSession { * generation path. Refresh via {@link AgentSession.setTitleSystemPrompt} when * the session cwd changes. */ #titleSystemPrompt: string | undefined; + #titleGenerationAbortController = new AbortController(); #toolChoiceQueue = new ToolChoiceQueue(); // Bash execution state #bashAbortControllers = new Set(); - #pendingBashMessages: BashExecutionMessage[] = []; + #pendingBashMessages: PendingBashMessage[] = []; + #bashSessionTarget!: BashSessionTarget; // Python execution state #evalAbortControllers = new Set(); @@ -1890,12 +2079,22 @@ export class AgentSession { #providerSessionId: string | undefined; #freshProviderSessionId: string | undefined; #inheritedProviderPromptCacheKey: string | undefined; + #autolearnCaptureAbortController: AbortController | undefined; + #autolearnCaptureTask: Promise | undefined; #isDisposed = false; // Extension system #extensionRunner: ExtensionRunner | undefined = undefined; + /** + * Backs `ctx.setInterval`/`setTimeout`/`clearTimer` for the runner-less + * command-context fallback (SDK embeddings with no extension runner). Lazily + * created; cleared on dispose alongside the runner's own timers (#5664). + */ + #fallbackExtensionTimers: ManagedTimers | undefined = undefined; #turnIndex = 0; #messageEndPersistenceTail: Promise = Promise.resolve(); #pendingMessageEndPersistence = new Map>(); + /** Async lifecycle handlers for visible advisor cards emitted outside the primary loop. */ + #pendingAdvisorCardEvents = new Set>(); #persistedMessageKeys: { anchor: string; keys: Set } | undefined; #skills: Skill[]; @@ -1907,6 +2106,7 @@ export class AgentSession { #mcpPromptCommands: LoadedCustomCommand[] = []; #skillsSettings: SkillsSettings | undefined; + #skillsReloadable: boolean; // Model registry for API key resolution #modelRegistry: ModelRegistry; @@ -1930,8 +2130,10 @@ export class AgentSession { #getLocalCalendarDate: () => string; #getMcpServerInstructions: (() => Map | undefined) | undefined; #setActiveToolNames: ((names: Iterable) => void) | undefined; + #ensureWriteRegistered: (() => Promise) | undefined; #disconnectOwnedMcpManager: (() => Promise) | undefined; - #requestedToolNames: ReadonlySet | undefined; + #presentationPinnedToolNames: ReadonlySet | undefined; + #runtimeSelectedToolNames: ReadonlySet | undefined; #baseSystemPrompt: string[]; #baseSystemPromptBeforeMemoryPromotion: string[] | undefined; /** @@ -1954,6 +2156,9 @@ export class AgentSession { #xdevRegistry: XdevRegistry | undefined; /** Names of discoverable tools currently mounted under `xd://` (dynamic mounts only, not built-in devices). */ #mountedXdevToolNames = new Set(); + /** Coalesced xd:// mount delta not yet announced to the model; delivered as a + * hidden notice alongside the next prompt (see {@link #notifyXdevMountDelta}). */ + #pendingXdevMountDelta: { added: Set; removed: Set } | undefined; // TTSR manager for time-traveling stream rules #ttsrManager: TtsrManager | undefined = undefined; @@ -2101,6 +2306,17 @@ export class AgentSession { * queue was consumed normally or a new turn already started. */ #drainStrandedQueuedMessages(): void { if (this.#abortInProgress) return; + // Session transitions (newSession/`/new`, compact, model-switch, session-switch, + // dispose) call #disconnectFromAgent() BEFORE `await abort()`, so abort's own + // finally lands here with no listener attached. Auto-resuming now would snapshot + // the still-old context (the transition hasn't reached agent.reset() yet), start a + // stale provider turn that races the reset, and — once reconnected — append its + // output to the fresh session (issue #5800). A disconnected session never owns the + // queue: the transition does. newSession/switchSession drop the queue (reset / + // clearAllQueues), so nothing survives; compaction preserves it and re-drains itself + // after #reconnectToAgent (see compact()'s finally); an explicit prompt flushes it + // in every case. + if (this.#unsubscribeAgent === undefined) return; // A concern steered into a resumed streaming run after a user interrupt can // strand at the turn tail (steered past the loop's final boundary poll). While // that interrupt's suppression is still in effect, reclaim such advisor steers @@ -2369,7 +2585,13 @@ export class AgentSession { const isPlanNudge = (m: AgentMessage): boolean => m.role === "custom" && m.customType === PREWALK_PLAN_MESSAGE_TYPE; for (let i = liveMessages.length - 1; i >= 0; i--) { - if (isPlanNudge(liveMessages[i])) liveMessages.splice(i, 1); + if (isPlanNudge(liveMessages[i])) { + // Interior removal on the live array: drop the scrubbed message from + // the convert/estimate caches so the next convert can't reuse a prefix + // that still carries its fragment (the array shrinks in place). + invalidateMessageCache(liveMessages[i]); + liveMessages.splice(i, 1); + } } const stateMessages = this.agent.state.messages; const filtered = stateMessages.filter(m => !isPlanNudge(m)); @@ -2399,6 +2621,24 @@ export class AgentSession { this.setPlanProposalHandler(title => this.#approvePlanYoloProposal(title)); } + /** Validate the active plan artifact and shape an `xd://propose` result for review-mode hosts. */ + async preparePlanForReview(title: string): Promise> { + const state = this.getPlanModeState(); + if (!state?.enabled) { + throw new ToolError("Plan mode is not active."); + } + const { planFilePath, title: resolvedTitle } = await resolveApprovedPlan({ + suppliedTitle: title, + statePlanFilePath: state.planFilePath, + readPlan: url => this.#readPlanFile(url), + listPlanFiles: () => this.#listPlanFiles(), + }); + return { + content: [{ type: "text", text: "Plan ready for review." }], + details: { planFilePath, title: resolvedTitle, planExists: true }, + }; + } + /** * Plan-proposal handler while PlanYolo's plan phase is active. Auto-approves * the instant the model writes the plan slug/title to `xd://propose` — no @@ -2419,15 +2659,20 @@ export class AgentSession { const { planFilePath, title: resolvedTitle } = await resolveApprovedPlan({ suppliedTitle: title, statePlanFilePath: state.planFilePath, - readPlan: url => this.#readPlanYoloFile(url), - listPlanFiles: () => this.#listPlanYoloFiles(), + readPlan: url => this.#readPlanFile(url), + listPlanFiles: () => this.#listPlanFiles(), }); + this.setPlanModeState(undefined); const previousTools = this.#planYoloPreviousTools; - if (previousTools) { - await this.setActiveToolsByName(previousTools); + try { + if (previousTools) { + await this.setActiveToolsByName(previousTools); + } + } catch (error) { + this.setPlanModeState(state); + throw error; } this.setPlanProposalHandler(null); - this.setPlanModeState(undefined); this.#planYolo = undefined; this.#planYoloPreviousTools = undefined; await this.setModelTemporary(planYolo.target, planYolo.thinkingLevel, { ephemeral: true }); @@ -2450,7 +2695,7 @@ export class AgentSession { }; } - async #readPlanYoloFile(planFilePath: string): Promise { + async #readPlanFile(planFilePath: string): Promise { const resolvedPath = planFilePath.startsWith("local:") ? resolveLocalUrlToPath(normalizeLocalScheme(planFilePath), this.#localProtocolOptions()) : resolveToCwd(planFilePath, this.sessionManager.getCwd()); @@ -2464,7 +2709,7 @@ export class AgentSession { /** `local://` URLs of plan files in the session-local root, newest first — * a fallback for `resolveApprovedPlan` when the agent dropped `extra.title`. */ - async #listPlanYoloFiles(): Promise { + async #listPlanFiles(): Promise { const localRoot = resolveLocalUrlToPath("local://", this.#localProtocolOptions()); try { const entries = await fs.promises.readdir(localRoot, { withFileTypes: true }); @@ -2485,6 +2730,11 @@ export class AgentSession { constructor(config: AgentSessionConfig) { this.agent = config.agent; this.sessionManager = config.sessionManager; + this.#bashSessionTarget = { + sessionId: this.sessionManager.getSessionId(), + refs: 0, + destination: { kind: "current", manager: this.sessionManager }, + }; this.settings = config.settings; this.#autoApprove = config.autoApprove === true; // Power assertions are taken per turn (see #beginInFlight); nothing acquired here. @@ -2502,6 +2752,13 @@ export class AgentSession { } else { this.#thinkingLevel = config.thinkingLevel; } + if (config.initialRetryFallback) { + this.#activeRetryFallback = { + ...config.initialRetryFallback, + lastAppliedFallbackThinkingLevel: this.configuredThinkingLevel(), + pinned: false, + }; + } if (config.prewalk) { this.#prewalk = config.prewalk; } @@ -2516,6 +2773,7 @@ export class AgentSession { this.#skills = config.skills ?? []; this.#skillWarnings = config.skillWarnings ?? []; this.#customCommands = config.customCommands ?? []; + this.#skillsReloadable = config.skillsReloadable ?? true; this.#skillsSettings = config.skillsSettings; this.#modelRegistry = config.modelRegistry; // Resolve the wire service-tier per request so the Fireworks Priority @@ -2534,7 +2792,8 @@ export class AgentSession { this.#toolRegistry = config.toolRegistry ?? new Map(); this.#createVibeTools = config.createVibeTools; this.#builtInToolNames = new Set(config.builtInToolNames ?? []); - this.#requestedToolNames = config.requestedToolNames; + this.#presentationPinnedToolNames = config.presentationPinnedToolNames; + this.#ensureWriteRegistered = config.ensureWriteRegistered; this.#transformContext = config.transformContext ?? (messages => messages); this.#transformProviderContext = config.transformProviderContext; this.#sideStreamFn = config.sideStreamFn ?? streamSimple; @@ -2586,7 +2845,18 @@ export class AgentSession { this.#advisorPrimaryTurnsCompleted++; if (this.#advisors.length > 0) { for (const a of this.#advisors) { - if (!a.runtime.disposed) a.runtime.onTurnEnd(messages, { willContinue: context?.willContinue }); + if (a.runtime.disposed) continue; + try { + a.runtime.onTurnEnd(messages, { willContinue: context?.willContinue }); + } catch (advisorErr) { + // CRITICAL boundary: NOTHING an advisor does may abort the + // primary agent's turn-end. A throwing advisor loses its + // delta; the primary continues untouched. + logger.warn("advisor onTurnEnd threw; delta dropped", { + advisor: a.name, + err: String(advisorErr), + }); + } } const syncBacklog = this.settings.get("advisor.syncBacklog"); if (syncBacklog !== "off") { @@ -2795,6 +3065,12 @@ export class AgentSession { slug = candidate; usedSlugs.add(slug); } + // Per-advisor toggle: skip disabled advisors but keep them in the + // status map so they show `○` rather than disappearing. + if (config.enabled === false) { + this.#advisorStatuses.set(slug, { name: config.name, status: "paused" }); + continue; + } // Resolve the advisor's model: an explicit `model` override wins; else the // `advisor` role chain. A model that fails to resolve skips just this advisor. @@ -2805,6 +3081,7 @@ export class AgentSession { model = resolved.model; thinkingLevel = concreteThinkingLevel(resolved.thinkingLevel); if (!model) { + this.#advisorStatuses.set(slug, { name: config.name, status: "no_model" }); if (emitWarnings) { this.emitNotice("warning", `Advisor "${config.name}": no model matched "${config.model}"`, "advisor"); } @@ -2813,6 +3090,7 @@ export class AgentSession { } else { const sel = resolveAdvisorRoleSelection(this.settings, this.#modelRegistry.getAvailable()); if (!sel) { + this.#advisorStatuses.set(slug, { name: config.name, status: "no_model" }); if (emitWarnings) { logger.debug("advisor enabled but no model assigned to the 'advisor' role; advisor inactive", { advisor: config.name, @@ -2836,6 +3114,11 @@ export class AgentSession { const requestedLevel = thinkingLevel ?? ThinkingLevel.Medium; const resolvedLevel = resolveThinkingLevelForModel(model, requestedLevel); const advisorThinkingLevel: ThinkingLevel = resolvedLevel ?? ThinkingLevel.Inherit; + // Record the status entry now (in roster order) so the Map's insertion + // order matches the configured roster even when earlier advisors were + // skipped as paused/no_model. The build loop overwrites this to "running" + // without changing insertion order. + this.#advisorStatuses.set(slug, { name: config.name, status: "running" }); descriptors.push({ config, name: config.name, @@ -2871,6 +3154,11 @@ export class AgentSession { if (!this.#advisorEnabled) return false; if (this.#agentKind !== "main" && !this.settings.get("advisor.subagents")) return false; + // Rebuild the status map from scratch so removed/renamed advisors don't + // leave stale entries. #resolveAdvisorRuntimeDescriptors populates every + // entry (`paused`/`no_model`/`running`) in roster order; the build loop + // below confirms `running` for successfully built advisors. + this.#advisorStatuses.clear(); const descriptors = this.#resolveAdvisorRuntimeDescriptors(true); // Advisor service tier (`tier.advisor`): "none" (default) runs the advisor @@ -2911,11 +3199,16 @@ export class AgentSession { const names = config.tools === undefined ? ADVISOR_DEFAULT_TOOL_NAMES : new Set(config.tools); const tools = (this.#advisorTools ?? []).filter(t => names.has(t.name)); + const advisorLoopTools: AgentTool[] = [adviseTool, ...tools]; + const advisorToolMap = new Map(); const availableAdvisorToolNames = new Set(); - availableAdvisorToolNames.add(adviseTool.name); - for (const tool of tools) { + for (const tool of advisorLoopTools) { availableAdvisorToolNames.add(tool.name); - if (tool.customWireName !== undefined) availableAdvisorToolNames.add(tool.customWireName); + advisorToolMap.set(tool.name, tool); + if (tool.customWireName !== undefined) { + availableAdvisorToolNames.add(tool.customWireName); + advisorToolMap.set(tool.customWireName, tool); + } } let quarantinedAdvisorOutput: string | undefined; let currentAdvisorInput = ""; @@ -2955,17 +3248,36 @@ export class AgentSession { // Codex request identity remains UUID-shaped while local labels keep the // `-advisor` suffix. const advisorPromptCacheKey = this.agent.promptCacheKey ?? advisorProviderSessionId; + // On the Cursor provider every tool runs server-side and is dispatched + // back through `cursorExecHandlers`; without this bridge the advisor's + // own tools (including the MCP `advise` tool) return `toolNotFound` and + // no advice is ever routed (issue #5680). Mirrors the primary agent's + // bridge (`sdk.ts`), scoped to this advisor's granted tool set. + // Cursor's native `delete` frame removes files directly, bypassing the + // tool map, so gate it on the advisor actually holding a file-mutating + // tool. A default read-only advisor (advise/read/grep/glob) never gets + // to delete workspace files it was never granted (issue #5680 review). + const advisorCanMutateFiles = advisorToolMap.has("write") || advisorToolMap.has("edit"); + if (advisorCanMutateFiles) availableAdvisorToolNames.add("delete"); + const advisorCursorExecHandlers = new CursorExecHandlers({ + cwd: this.sessionManager.getCwd(), + getCwd: () => this.sessionManager.getCwd(), + tools: advisorToolMap, + allowNativeDelete: advisorCanMutateFiles, + }); const advisorAgent = new Agent({ initialState: { systemPrompt, model: advisorModel, thinkingLevel: toReasoningEffort(advisorThinkingLevel), - tools: [adviseTool, ...tools], + tools: advisorLoopTools, }, appendOnlyContext, sessionId: advisorProviderSessionId, promptCacheKey: advisorPromptCacheKey, providerSessionState: this.#providerSessionState, + cursorExecHandlers: advisorCursorExecHandlers, + cwdResolver: () => this.sessionManager.getCwd(), preferWebsockets: this.#preferWebsockets, getApiKey: requestModel => this.#modelRegistry.resolver(requestModel, advisorProviderSessionId), streamFn: this.#advisorStreamFn, @@ -3036,33 +3348,30 @@ export class AgentSession { maintainContext: incomingTokens => this.#maintainAdvisorContext(advisorRef, incomingTokens), obfuscator: this.#obfuscator, beginAdvisorUpdate: () => advisorRef.emissionGuard.beginUpdate(), - onTurnError: async error => { - // Mirror the auth-gateway's usage-limit remedy: the in-stream a/b/c - // auth retry rotates through siblings within one request but never - // blocks the LAST failing credential, so without this the advisor - // re-picks the same exhausted account every retry. Usage limits - // only — other failures keep the plain retry/notify path (never - // suspect-mark a credential on a transient advisor error). - const message = error instanceof Error ? error.message : String(error); - if (!isUsageLimitOutcome(extractHttpStatusFromError(error), message)) return; - await this.#modelRegistry.authStorage.markUsageLimitReached( - advisorModel.provider, - advisorProviderSessionId, - { - retryAfterMs: extractRetryHint(undefined, message), - baseUrl: advisorModel.baseUrl, - modelId: advisorModel.id, - }, - ); + onTurnError: (error, failedMessages) => this.#recoverAdvisorTurn(advisorRef, error, failedMessages), + onTurnSuccess: async () => { + const fallback = advisorRef.retryFallback; + if (!advisorRef.retryFallbackPendingSuccess || !fallback) return; + advisorRef.retryFallbackPendingSuccess = false; + await this.#emitSessionEvent({ + type: "retry_fallback_succeeded", + model: formatRetryFallbackSelector(advisorRef.agent.state.model, advisorRef.thinkingLevel), + role: fallback.role, + }); }, notifyFailure: error => { + this.#advisorStatuses.set(slug, { name: advisorName, status: "error" }); const message = error instanceof Error ? error.message : String(error); this.emitNotice( "warning", - `Advisor${slug ? ` "${advisorName}"` : ""} unavailable for ${formatModelString(advisorModel)}: ${message}`, + `Advisor${slug ? ` "${advisorName}"` : ""} unavailable for ${formatModelString(advisorAgent.state.model)}: ${message}`, "advisor", ); }, + notifyQuotaExhausted: () => { + this.#advisorStatuses.set(slug, { name: advisorName, status: "quota_exhausted" }); + this.emitNotice("warning", `Advisor "${advisorName}" quota exhausted — pausing until reset.`, "advisor"); + }, }); const advisorRef: ActiveAdvisor = { @@ -3076,10 +3385,13 @@ export class AgentSession { recorderClosed: Promise.resolve(), model: advisorModel, thinkingLevel: advisorThinkingLevel, + providerSessionId: advisorProviderSessionId, + retryFallbackPendingSuccess: false, signature, }; this.#attachAdvisorRecorderFeed(advisorRef); if (seedToCurrent) runtime.seedTo(this.agent.state.messages.length); + this.#advisorStatuses.set(slug, { name: advisorName, status: "running" }); this.#advisors.push(advisorRef); } @@ -3110,11 +3422,11 @@ export class AgentSession { /** * Route one accepted advice note from `advisor` to the primary. Concern and * blocker interrupt the running agent through the steering channel; once the - * loop has yielded, `triggerTurn` resumes it. If the loop already ended with a - * terminal text answer and no queued work remains, the note is preserved as an - * advisor card instead of waking a duplicate completion turn. After a deliberate - * user interrupt auto-resume is suppressed while idle/unwinding (the note - * becomes a preserved card re-entering on resume); a live-streaming turn is + * loop has yielded, `triggerTurn` resumes it. After a terminal text answer with + * no queued work, a concern is preserved as a visible advisor card, while a + * blocker wakes the primary to acknowledge work it handed off incorrectly. + * After a deliberate user interrupt auto-resume is suppressed while idle/unwinding + * (the note becomes a preserved card re-entering on resume); a live-streaming turn is * steered in directly. A plain nit always rides the non-interrupting YieldQueue * aside. Suppression by the per-advisor emission guard drops the note silently — * the model still saw `Recorded.`, so it isn't tempted to rephrase the same note @@ -3144,6 +3456,7 @@ export class AgentSession { const channel = resolveAdvisorDeliveryChannel({ severity, autoResumeSuppressed: this.#advisorAutoResumeSuppressed, + preserveOnly: this.#preserveAdvisorAdvice, // Key on the live agent-core loop, not session `isStreaming` (which also // counts `#promptInFlightCount` during post-turn unwind). Only a running // loop consumes a steer at its next boundary. @@ -3171,10 +3484,19 @@ export class AgentSession { }); return; } - this.#recordAdvisorInterruptDelivered(); - if (this.#planModeState?.enabled) { - // Plan mode: record advice visibly in context but never wake an - // autonomous turn — only user-driven turns converge on ask/resolve. + // A steered interrupting note only continues the run when the session can + // actually start (or is already running) a turn. Two idle cases cannot, so + // `sendCustomMessage({ triggerTurn: true })` would silently bury the card in + // `#pendingNextTurnMessages` until the next user prompt — strictly worse than + // the visible preserved card. Preserve instead: + // - Plan mode: only user-driven turns converge on ask/resolve. + // - ACP bridges with `deferAgentInitiatedTurns`: the client cannot show an + // agent-initiated turn as busy, so idle triggers are refused (#5628 review). + const cannotAutoTrigger = + !this.agent.state.isStreaming && + this.#clientBridge?.deferAgentInitiatedTurns === true && + !this.#allowAcpAgentInitiatedTurns; + if (this.#planModeState?.enabled || cannotAutoTrigger) { this.#preserveAdvisorCard({ role: "custom", customType: "advisor", @@ -3186,6 +3508,11 @@ export class AgentSession { }); return; } + // Arm the post-interrupt immune window only now that a turn is actually + // being steered/triggered. A merely preserved card never interrupts, so + // arming earlier would downgrade the next `advisor.immuneTurns` worth of + // real concerns/blockers to skip-idle-flush asides (#5628 review). + this.#recordAdvisorInterruptDelivered(); void this.sendCustomMessage( { customType: "advisor", content, display: true, attribution: "agent", details }, { deliverAs: "steer", triggerTurn: true }, @@ -3227,6 +3554,163 @@ export class AgentSession { }); } + /** Switch one advisor model while preserving its context and effort invariants. */ + #setAdvisorModel(advisor: ActiveAdvisor, model: Model, requestedThinkingLevel: ThinkingLevel): ThinkingLevel { + const resolvedThinkingLevel = resolveThinkingLevelForModel(model, requestedThinkingLevel); + const nextThinkingLevel = resolvedThinkingLevel ?? ThinkingLevel.Inherit; + advisor.agent.setModel(model); + advisor.agent.setThinkingLevel(toReasoningEffort(nextThinkingLevel)); + advisor.agent.setDisableReasoning(shouldDisableReasoning(nextThinkingLevel)); + advisor.agent.appendOnlyContext?.invalidateForModelChange(); + advisor.model = model; + advisor.thinkingLevel = nextThinkingLevel; + return nextThinkingLevel; + } + + /** Restore an advisor's configured primary once its fallback cooldown expires. */ + async #maybeRestoreAdvisorRetryFallbackPrimary(advisor: ActiveAdvisor): Promise { + const fallback = advisor.retryFallback; + if (!fallback || this.#getRetryFallbackRevertPolicy() !== "cooldown-expiry") return; + + const originalSelector = parseRetryFallbackSelector(fallback.originalSelector, this.#modelRegistry); + if (!originalSelector) { + advisor.retryFallback = undefined; + advisor.retryFallbackPendingSuccess = false; + return; + } + const currentSelector = formatRetryFallbackSelector(advisor.agent.state.model, advisor.thinkingLevel); + if (currentSelector === originalSelector.raw) { + if (!this.#isRetryFallbackSelectorSuppressed(originalSelector)) { + advisor.retryFallback = undefined; + advisor.retryFallbackPendingSuccess = false; + } + return; + } + if (this.#isRetryFallbackSelectorSuppressed(originalSelector)) return; + + const resolvedPrimary = resolveModelOverride([originalSelector.raw], this.#modelRegistry, this.settings); + const primaryModel = + resolvedPrimary.model ?? this.#modelRegistry.find(originalSelector.provider, originalSelector.id); + if (!primaryModel) return; + const apiKey = await this.#modelRegistry.getApiKey(primaryModel, advisor.providerSessionId); + if (!apiKey) return; + + const thinkingToApply = + advisor.thinkingLevel === fallback.lastAppliedThinkingLevel + ? fallback.originalThinkingLevel + : advisor.thinkingLevel; + this.#setAdvisorModel(advisor, primaryModel, thinkingToApply); + this.settings.getStorage()?.recordModelUsage(formatModelStringWithRouting(primaryModel)); + advisor.retryFallback = undefined; + advisor.retryFallbackPendingSuccess = false; + } + + /** + * Apply the advisor's configured provider-failure fallback chain after + * same-provider credential rotation has no usable sibling. + */ + async #recoverAdvisorTurn( + advisor: ActiveAdvisor, + error: unknown, + failedMessages: readonly AgentMessage[], + ): Promise { + if (error instanceof AdvisorOutputQuarantinedError) return false; + + const failedMessage = failedMessages.findLast( + (message): message is AssistantMessage => message.role === "assistant", + ); + if (failedMessage?.stopReason !== "error") { + // Stream setup can reject before any assistant turn is recorded (e.g. + // an HTTP 429 thrown from prompt()); classify the raw error so a + // structural usage limit still marks the exhausted credential. + const message = error instanceof Error ? error.message : String(error); + if (!AIError.isUsageLimit(error) && !isUsageLimitOutcome(extractHttpStatusFromError(error), message)) { + return false; + } + const currentModel = advisor.agent.state.model; + const outcome = await this.#modelRegistry.authStorage.markUsageLimitReached( + currentModel.provider, + advisor.providerSessionId, + { + retryAfterMs: extractRetryHint(undefined, message), + baseUrl: currentModel.baseUrl, + modelId: currentModel.id, + }, + ); + return outcome.switched; + } + if (failedMessage.content.some(block => block.type === "toolCall")) return false; + + const currentModel = advisor.agent.state.model; + const message = failedMessage.errorMessage ?? (error instanceof Error ? error.message : String(error)); + const errorId = AIError.classifyMessage({ + api: currentModel.api, + errorId: failedMessage.errorId, + errorMessage: message, + errorStatus: failedMessage.errorStatus, + }); + if (AIError.is(errorId, AIError.Flag.Abort) || AIError.is(errorId, AIError.Flag.UserInterrupt)) return false; + if (AIError.isContextOverflow(failedMessage, currentModel.contextWindow ?? 0)) return false; + + const currentSelector = formatRetryFallbackSelector(currentModel, advisor.thinkingLevel); + + const retryAfterMs = extractRetryHint(undefined, message); + if ( + AIError.is(errorId, AIError.Flag.UsageLimit) || + isUsageLimitOutcome(extractHttpStatusFromError(error), message) + ) { + const outcome = await this.#modelRegistry.authStorage.markUsageLimitReached( + currentModel.provider, + advisor.providerSessionId, + { + retryAfterMs, + baseUrl: currentModel.baseUrl, + modelId: currentModel.id, + }, + ); + if (outcome.switched) return true; + } + + const retrySettings = this.settings.getGroup("retry"); + if (!retrySettings.enabled || !retrySettings.modelFallback) return false; + const role = advisor.retryFallback?.role ?? this.#resolveRetryFallbackRole(currentSelector, currentModel); + if (!role || this.#findRetryFallbackCandidates(role, currentSelector, currentModel).length === 0) return false; + + this.#noteRetryFallbackCooldown(currentSelector, retryAfterMs, message); + for (const selector of this.#findRetryFallbackCandidates(role, currentSelector, currentModel)) { + if (this.#isRetryFallbackSelectorSuppressed(selector)) continue; + const resolved = resolveModelOverride([selector.raw], this.#modelRegistry, this.settings); + const candidate = resolved.model ?? this.#modelRegistry.find(selector.provider, selector.id); + if (!candidate || modelsAreEqual(candidate, currentModel)) continue; + const apiKey = await this.#modelRegistry.getApiKey(candidate, advisor.providerSessionId); + if (!apiKey) continue; + + const originalThinkingLevel = advisor.thinkingLevel; + const requestedThinkingLevel = selector.thinkingLevel ?? originalThinkingLevel; + const nextThinkingLevel = this.#setAdvisorModel(advisor, candidate, requestedThinkingLevel); + if (advisor.retryFallback) { + advisor.retryFallback.lastAppliedThinkingLevel = nextThinkingLevel; + } else { + advisor.retryFallback = { + role, + originalSelector: currentSelector, + originalThinkingLevel, + lastAppliedThinkingLevel: nextThinkingLevel, + }; + } + advisor.retryFallbackPendingSuccess = true; + this.settings.getStorage()?.recordModelUsage(formatModelStringWithRouting(candidate)); + await this.#emitSessionEvent({ + type: "retry_fallback_applied", + from: currentSelector, + to: selector.raw, + role, + }); + return true; + } + return false; + } + async #promoteAdvisorContextModel(advisor: ActiveAdvisor, currentModel: Model): Promise { const promotionSettings = this.settings.getGroup("contextPromotion"); if (!promotionSettings.enabled) return false; @@ -3239,10 +3723,7 @@ export class AgentSession { // keeps its suffix across a promotion); only the model changes. const advisorThinkingLevel = advisor.thinkingLevel; try { - advisor.agent.setModel(targetModel); - advisor.agent.setThinkingLevel(toReasoningEffort(advisorThinkingLevel)); - advisor.agent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel)); - advisor.agent.appendOnlyContext?.invalidateForModelChange(); + this.#setAdvisorModel(advisor, targetModel, advisorThinkingLevel); logger.debug("Advisor context promotion switched model on overflow", { advisor: advisor.name, from: `${currentModel.provider}/${currentModel.id}`, @@ -3261,6 +3742,7 @@ export class AgentSession { } async #maintainAdvisorContext(advisor: ActiveAdvisor, incomingTokens: number): Promise { + await this.#maybeRestoreAdvisorRetryFallbackPrimary(advisor); const agent = advisor.agent; const compactionSettings = this.settings.getGroup("compaction"); @@ -3272,10 +3754,23 @@ export class AgentSession { if (contextWindow <= 0) return false; const messages = agent.state.messages; - let contextTokens = incomingTokens; + const estimateOptions = { excludeEncryptedReasoning: true } as const; + let storedConversationTokens = 0; for (const message of messages) { - contextTokens += estimateTokens(message); + storedConversationTokens += estimateTokens(message, estimateOptions); } + // Provider usage (including cache reads and generated output) is the + // trustworthy anchor for accumulated context. Add only the trailing incoming + // delta to that arm. Floor it by a full local estimate — fixed advisor system + // prompt, tool schemas, stored messages, and incoming delta — so provider + // under-reporting or payload transforms cannot suppress maintenance. + const providerContextTokens = this.#estimateAdvisorContextTokens(messages) + incomingTokens; + const localContextTokens = + countTokens(agent.state.systemPrompt) + + estimateToolSchemaTokens(agent.state.tools) + + storedConversationTokens + + incomingTokens; + const contextTokens = compactionContextTokens(providerContextTokens, localContextTokens); if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) { return false; @@ -3299,6 +3794,7 @@ export class AgentSession { const timestamp = String(message.timestamp || Date.now()); if (message.role === "compactionSummary") { + const advisorSummary = message as AdvisorCompactionSummaryMessage; return { type: "compaction", id, @@ -3306,9 +3802,7 @@ export class AgentSession { timestamp, summary: message.summary, shortSummary: message.shortSummary, - firstKeptEntryId: - (message as CompactionSummaryMessage & { firstKeptEntryId?: string }).firstKeptEntryId || - `msg-${i + 1}`, + firstKeptEntryId: advisorSummary.firstKeptEntryId || `msg-${i + 1}`, tokensBefore: message.tokensBefore, } satisfies CompactionEntry; } @@ -3402,11 +3896,15 @@ export class AgentSession { const firstKeptEntryId = compactResult.firstKeptEntryId; const tokensBefore = compactResult.tokensBefore; - // Rebuild messages with the compaction summary + // The retained messages still carry provider usage from before this + // compaction. Record their exact array boundary on the in-memory summary so + // only assistants appended afterward can become the next usage anchor. + const advisorUsageAnchorStartIndex = preparation.recentMessages.length + 1; const summaryMessage = { ...createCompactionSummaryMessage(summary, tokensBefore, new Date().toISOString(), shortSummary), firstKeptEntryId, - } as CompactionSummaryMessage & { firstKeptEntryId?: string }; + advisorUsageAnchorStartIndex, + } satisfies AdvisorCompactionSummaryMessage; agent.replaceMessages([summaryMessage, ...preparation.recentMessages]); return false; @@ -3758,30 +4256,58 @@ export class AgentSession { return queued; } + /** + * Orders subscriber fan-out across concurrent `#emitSessionEvent` calls. + * Extension emits only await when the event type has handlers, so an event + * with no handlers could otherwise overtake an earlier event still inside + * its extension emit — an instant refusal delivered its assistant + * `message_end` to the TUI before its own `message_start`, skipping the + * turn-ending error render entirely. + */ + #subscriberEmitGate: Promise = Promise.resolve(); + async #emitSessionEvent(event: AgentSessionEvent): Promise { if (event.type === "message_update") { this.#emit(event); void this.#queueExtensionEvent(event); return; } - await this.#emitExtensionEvent(event); - // Hold the wire-level agent_end until in-flight prompts unwind. Subscribers - // (rpc-mode, ACP, Cursor) treat agent_end as the "session is idle" signal; - // emitting while #promptInFlightCount > 0 lets a client fire its next - // `prompt` into a session that still reports isStreaming === true. Flush - // happens in #endInFlight / #resetInFlight. A later agent_end (e.g. from - // an auto-compaction turn that starts before the original prompt unwinds) - // supersedes the pending one, which is what subscribers want — they only - // care about the final settle. - if (event.type === "agent_end" && this.#promptInFlightCount > 0) { - this.#pendingAgentEndEmit = event; - return; + // Take a FIFO ticket before the extension emit: extension deliveries for + // consecutive events still run concurrently, but subscriber fan-out waits + // for every earlier event's fan-out (or deferral) to happen first. + const previousGate = this.#subscriberEmitGate; + const { promise: gate, resolve: releaseGate } = Promise.withResolvers(); + this.#subscriberEmitGate = gate; + try { + await this.#emitExtensionEvent(event); + await previousGate; + // Hold the wire-level agent_end until in-flight prompts unwind. Subscribers + // (rpc-mode, ACP, Cursor) treat agent_end as the "session is idle" signal; + // emitting while #promptInFlightCount > 0 lets a client fire its next + // `prompt` into a session that still reports isStreaming === true. Flush + // happens in #endInFlight / #resetInFlight. A later agent_end (e.g. from + // an auto-compaction turn that starts before the original prompt unwinds) + // supersedes the pending one, which is what subscribers want — they only + // care about the final settle. + if (event.type === "agent_end" && this.#promptInFlightCount > 0) { + this.#pendingAgentEndEmit = event; + return; + } + this.#emit(event); + } finally { + releaseGate(); } - this.#emit(event); } // Track last assistant message for auto-compaction check #lastAssistantMessage: AssistantMessage | undefined = undefined; + /** + * Classifier-refusal turn pruned from active context at settle (#3591). + * Retained until the next run starts so post-settle readers + * ({@link getLastAssistantMessage}: print mode, task executor) still see + * the terminal error instead of a silently successful-looking state. + */ + #prunedTerminalRefusal: AssistantMessage | undefined = undefined; /** Internal handler for agent events - shared by subscribe and reconnect. * @@ -3799,7 +4325,12 @@ export class AgentSession { * everything it schedules — settles. */ #handleAgentEvent = async (event: AgentEvent): Promise => { if (event.type !== "agent_end") { - return this.#processAgentEvent(event); + const processing = this.#processAgentEvent(event); + if ((event.type === "message_start" || event.type === "message_end") && isAdvisorCard(event.message)) { + this.#pendingAdvisorCardEvents.add(processing); + void processing.finally(() => this.#pendingAdvisorCardEvents.delete(processing)).catch(() => {}); + } + return processing; } const { promise, resolve } = Promise.withResolvers(); this.#trackPostPromptTask(promise); @@ -3962,7 +4493,7 @@ export class AgentSession { } const skipPersistedRewindResult = message.role === "toolResult" && - message.toolName === "rewind" && + semanticToolResult(message.toolName, message)?.toolName === "rewind" && this.#rewoundToolResultIds.delete(message.toolCallId); if (!skipPersistedRewindResult) { this.#appendSessionMessage(message); @@ -4041,6 +4572,11 @@ export class AgentSession { } #processAgentEvent = async (event: AgentEvent): Promise => { + // A fresh run supersedes the previously settled (and pruned) refusal + // turn: state-based lookups take over again. + if (event.type === "agent_start") { + this.#prunedTerminalRefusal = undefined; + } // Step the mid-run todo counter synchronously, BEFORE any await in this // handler. The agent loop's next-turn `getAsideMessages` poll can run // before queued microtasks drain, so `#takeMidRunTodoNudge` MUST see the @@ -4059,6 +4595,19 @@ export class AgentSession { } else if (!isError && MID_RUN_TODO_NUDGE_MUTATING_TOOLS[toolName]) { this.#mutationsSinceLastTodoTouch++; } + // A tool actually ran. Clear the post-reminder suppression synchronously + // too: the settle check (`#checkTodoCompletion` in agent_end maintenance) + // can otherwise read the stale flag when a tool result and the terminal + // stop land in the same tick, swallowing the earned re-escalation. + this.#todoReminderAwaitingProgress = false; + } + // Track the settled assistant turn synchronously as well: agent_end + // maintenance reads `#lastAssistantMessage`, and when a turn's events all + // land in one tick its handler can run before this handler's post-emit + // bookkeeping — leaving maintenance looking at the previous (e.g. + // toolUse) assistant message and skipping settle-only work. + if (event.type === "message_end" && event.message.role === "assistant") { + this.#lastAssistantMessage = event.message; } // Plan-mode internal transition: stamp `SILENT_ABORT_MARKER` on the // persisted message BEFORE the obfuscator's display-side copy below. @@ -4265,9 +4814,7 @@ export class AgentSession { } // Other message types (bashExecution, compactionSummary, branchSummary) are persisted elsewhere - // Track assistant message for auto-compaction (checked on agent_end) if (event.message.role === "assistant") { - this.#lastAssistantMessage = event.message; const assistantMsg = event.message as AssistantMessage; // Fold this turn's timing into per-model perf aggregates (drives the // /models TPS/TTFT display). Errored turns measure nothing; aborted @@ -4333,29 +4880,25 @@ export class AgentSession { } } if (event.message.role === "toolResult") { - const { toolName, toolCallId, details, isError, content } = event.message as { - toolCallId?: string; - toolName?: string; - details?: { op?: string; path?: string; phases?: TodoPhase[]; report?: string; startedAt?: string }; - isError?: boolean; - content?: Array; - }; - // A tool actually ran. Clear the post-reminder suppression: the agent did - // productive work in response to the prior nudge, so the next text-only stop - // is allowed to escalate to the next reminder if todos remain incomplete. - this.#todoReminderAwaitingProgress = false; + const { toolName, toolCallId, isError, content } = event.message; + const details = isRecord(event.message.details) ? event.message.details : undefined; + const semanticResult = semanticToolResult(toolName, event.message); + const semanticDetails = isRecord(semanticResult?.details) ? semanticResult.details : undefined; // Invalidate streaming edit cache when edit tool completes to prevent stale data - if (toolName === "edit" && details?.path) { - this.#invalidateFileCacheForPath(details.path); + const editedPath = details ? getStringProperty(details, "path") : undefined; + if (toolName === "edit" && editedPath) { + this.#invalidateFileCacheForPath(editedPath); } - if (toolName === "todo" && !isError && Array.isArray(details?.phases)) { - this.setTodoPhases(details.phases); + // TodoTool commits its state during execute. Replaying the result after + // awaited event fan-out can overwrite a newer call from the same batch. + const phases = details?.phases; + if (toolName === "todo" && !isError && details && Array.isArray(phases) && phases.every(isTodoPhase)) { if (this.#isTodoInitResult(details, toolCallId)) { this.#scheduleReplanTitleRefresh(); } } if (toolName === "todo" && isError) { - const errorText = content?.find(part => part.type === "text")?.text; + const errorText = content.find(part => part.type === "text")?.text; const reminderText = [ "", "todo failed, so todo progress is not visible to the user.", @@ -4373,18 +4916,19 @@ export class AgentSession { { deliverAs: "nextTurn" }, ); } - if (toolName === "checkpoint" && !isError) { + if (semanticResult?.toolName === "checkpoint" && !isError) { const checkpointEntryId = this.sessionManager.getEntries().at(-1)?.id ?? null; this.#checkpointState = { checkpointMessageCount: this.agent.state.messages.length, checkpointEntryId, - startedAt: details?.startedAt ?? new Date().toISOString(), + startedAt: + (semanticDetails && stringProperty(semanticDetails, "startedAt")) ?? new Date().toISOString(), }; this.#pendingRewindReport = undefined; this.#lastCompletedRewind = undefined; } - if (toolName === "rewind" && !isError && this.#checkpointState) { - const detailReport = typeof details?.report === "string" ? details.report.trim() : ""; + if (semanticResult?.toolName === "rewind" && !isError && this.#checkpointState) { + const detailReport = semanticDetails ? (stringProperty(semanticDetails, "report")?.trim() ?? "") : ""; const textReport = content?.find(part => part.type === "text")?.text?.trim() ?? ""; const report = detailReport || textReport; if (report.length > 0) { @@ -4397,8 +4941,11 @@ export class AgentSession { // Check auto-retry and auto-compaction after agent completes if (event.type === "agent_end") { const settledMessages = this.agent.state.messages; - const emitAgentEndNotification = async () => { - await this.#emitAgentEndNotification(settledMessages); + // TTSR retry work runs concurrently and clears the live flag before + // maintenance can emit agent_end, so preserve the state at settle entry. + const ttsrAbortPendingAtAgentEnd = this.#ttsrAbortPending; + const emitAgentEndNotification = async (options?: { willContinue?: boolean }) => { + await this.#emitAgentEndNotification(settledMessages, options); }; const usage = this.getSessionStats().tokens; await this.#goalRuntime.onAgentEnd({ @@ -4445,6 +4992,12 @@ export class AgentSession { }; maintenanceRoute("entered"); + // Surface provider stream failures in the main log. The routing trace + // above is debug-only and drops the error fields, so a session dying + // repeatedly on provider errors otherwise leaves no actionable trace + // outside the session transcript (issue #6177). + logProviderTurnError(msg); + // Invalidate GitHub Copilot credentials on auth failure so stale tokens // aren't reused on the next request if ( @@ -4500,7 +5053,7 @@ export class AgentSession { // active-goal threshold pre-empt below. if (await this.#handleEmptyAssistantStop(msg)) { maintenanceRoute("empty-stop-handled"); - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } @@ -4523,30 +5076,34 @@ export class AgentSession { automaticContinuationBlocked: compactionResult.automaticContinuationBlocked === true, }); this.#resolveRetry(); - await emitAgentEndNotification(); + await emitAgentEndNotification( + compactionResult.continuationScheduled ? { willContinue: true } : undefined, + ); return; } } if (await this.#handleUnexpectedAssistantStop(msg)) { maintenanceRoute("unexpected-stop-handled"); - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } if (this.#isRetryableReasonlessAbort(msg)) { const didRetry = await this.#handleRetryableError(msg, { allowModelFallback: false }); if (didRetry) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } } - // A deliberate abort should settle the current turn, not trigger queued continuations. + // A deliberate abort should settle the current turn, not trigger queued + // continuations — except TTSR self-repair, which already scheduled a + // hidden retry while #ttsrAbortPending is still true. if (msg.stopReason === "aborted") { this.#resolveRetry(); this.#resetSessionStopContinuationState(); - await emitAgentEndNotification(); + await emitAgentEndNotification(ttsrAbortPendingAtAgentEnd ? { willContinue: true } : undefined); return; } // Fireworks Fast variants degrade to their base model on a failed turn — @@ -4555,14 +5112,18 @@ export class AgentSession { if (this.#isFireworksFastFallbackEligible(msg)) { const didRetry = await this.#handleRetryableError(msg, { fireworksFastFallback: true }); if (didRetry) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } } - if (this.#isRetryableError(msg)) { - const didRetry = await this.#handleRetryableError(msg); + const resumeCursorStreamStall = this.#canResumeCursorStreamStall(msg); + if (resumeCursorStreamStall || this.#isRetryableError(msg)) { + const didRetry = await this.#handleRetryableError( + msg, + resumeCursorStreamStall ? { preserveFailedTurn: true } : undefined, + ); if (didRetry) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } } else if (this.#isHardErrorFallbackEligible(msg)) { @@ -4573,17 +5134,30 @@ export class AgentSession { // backoff-retry of the failing model) when no switch happens. const didRetry = await this.#handleRetryableError(msg, { hardErrorFallback: true }); if (didRetry) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } } // Classifier refusals are persisted-skipped above; also prune the trailing // stub from active context so the next turn's prompt does not replay it. + // Keep a reference for post-settle readers (print mode, task executor via + // getLastAssistantMessage) — pruning made the terminal error invisible to + // anything inspecting agent state after prompt() resolved. // Fall through to the standard error tail so `session_stop` hooks (block, // continue, telemetry) still fire — matching the pre-fix flow for // `stopReason === "error"`. if (this.#isClassifierRefusal(msg)) { + this.#prunedTerminalRefusal = msg; this.#removeAssistantMessageFromActiveContext(msg); + } else if (!AIError.isContextOverflow(msg, this.model?.contextWindow ?? 0)) { + // No retry, fallback, or compaction continuation fired: this errored + // turn ends the run. #persistSessionMessageIfMissing dropped it as an + // empty error turn, so record it here — otherwise the JSONL stops at + // the last tool result and the provider's errorMessage is lost (#6249). + // Idempotent and a no-op for non-empty turns. Content-less overflow + // rejections stay live-UI only per the auto-compaction progress guard: + // persisting one would replay an empty assistant turn on reload. + await this.#persistTerminalEmptyErrorTurn(msg); } this.#resolveRetry(); @@ -4612,22 +5186,22 @@ export class AgentSession { compactionResult.continuationScheduled || compactionResult.automaticContinuationBlocked ) { - await emitAgentEndNotification(); + await emitAgentEndNotification(compactionResult.continuationScheduled ? { willContinue: true } : undefined); return; } if (msg.stopReason !== "error") { if (this.#enforceRewindBeforeYield()) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } const planModeContinuationScheduled = await this.#enforcePlanModeDecisionAtSettle(); if (planModeContinuationScheduled) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } const todoContinuationScheduled = await this.#checkTodoCompletion(msg); if (todoContinuationScheduled) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } } @@ -4637,11 +5211,11 @@ export class AgentSession { // the session is fully idle (the todo reminder above defers the same // way inside #checkTodoCompletion). if (this.#hasPendingAsyncWake()) { - await emitAgentEndNotification(); + await emitAgentEndNotification({ willContinue: true }); return; } - await this.#emitSessionStopEvent(settledMessages, msg); - await emitAgentEndNotification(); + const sessionStopWillContinue = await this.#emitSessionStopEvent(settledMessages, msg); + await emitAgentEndNotification(sessionStopWillContinue ? { willContinue: true } : undefined); } }; @@ -4771,7 +5345,28 @@ export class AgentSession { ); } - #scheduleAutoContinuePrompt(generation: number): void { + #scheduleCompactionContinuation(options: { + generation: number; + autoContinue: boolean; + terminalTextAnswer: boolean; + suppressContinuation: boolean; + }): boolean { + if (options.suppressContinuation) return false; + if (this.agent.hasQueuedMessages()) { + this.#scheduleAgentContinue({ + delayMs: 100, + generation: options.generation, + shouldContinue: () => this.agent.hasQueuedMessages(), + }); + return true; + } + if (!options.autoContinue) return false; + const activeGoal = this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active"; + if (options.terminalTextAnswer && !activeGoal) return false; + return this.#scheduleAutoContinuePrompt(options.generation); + } + + #scheduleAutoContinuePrompt(generation: number): boolean { const continuePrompt = async () => { // Compaction summarizes away the first-message eager preludes, so re-assert the // delegate-via-tasks / phased-todo reminders on this auto-resumed turn. This runs @@ -4796,10 +5391,18 @@ export class AgentSession { async signal => { await Promise.resolve(); if (signal.aborted) return; + if (this.agent.hasQueuedMessages()) { + this.#scheduleAgentContinue({ + generation, + shouldContinue: () => this.agent.hasQueuedMessages(), + }); + return; + } await continuePrompt(); }, { generation }, ); + return true; } async #cancelPostPromptTasks(): Promise { @@ -5585,10 +6188,9 @@ export class AgentSession { // `local://` URLs (e.g. local://PLAN.md for plan-mode) resolve to a real // on-disk artifacts path; pre-caching works as long as we ask the - // local-protocol handler. Other internal-scheme URLs (agent://, skill://, - // rule://, mcp://, artifact://) have no stable filesystem representation; - // skip pre-cache entirely for those — the edit tool itself will reject - // them through its normal dispatch path. + // local-protocol handler. Other internal-scheme URLs have no local + // filesystem representation; skip pre-cache entirely for those — the + // edit tool itself will reject them through its normal dispatch path. const resolvedPath = this.#resolveSessionFsPath(path); if (resolvedPath === undefined) return undefined; @@ -5688,9 +6290,8 @@ export class AgentSession { * - `local://` URLs route through the local-protocol handler so they map * onto the session's on-disk artifacts directory; pre-caching, ENOENT * handling, and post-edit invalidation all work normally. - * - Other internal-scheme URLs (agent://, skill://, rule://, mcp://, - * artifact://) have no stable filesystem path; this returns `undefined` - * so callers skip filesystem-only operations. + * - Other internal-scheme URLs have no local filesystem path; this returns + * `undefined` so callers skip filesystem-only operations. * - Cwd-relative and absolute paths resolve via `resolveToCwd`. */ #resolveSessionFsPath(filePath: string): string | undefined { @@ -5698,15 +6299,7 @@ export class AgentSession { if (normalized.startsWith("local:")) { return resolveLocalUrlToPath(normalized, this.#localProtocolOptions()); } - if ( - normalized.startsWith("agent://") || - normalized.startsWith("skill://") || - normalized.startsWith("rule://") || - normalized.startsWith("mcp://") || - normalized.startsWith("artifact://") - ) { - return undefined; - } + if (isInternalUrlPath(normalized)) return undefined; return resolveToCwd(normalized, this.sessionManager.getCwd()); } @@ -5870,15 +6463,26 @@ export class AgentSession { return undefined; } - async #emitAgentEndNotification(messages: AgentMessage[]): Promise { - await this.#extensionRunner?.emit({ type: "agent_end", messages }); + async #emitAgentEndNotification(messages: AgentMessage[], options?: { willContinue?: boolean }): Promise { + await this.#extensionRunner?.emit({ + type: "agent_end", + messages, + willContinue: options?.willContinue, + }); } + /** @returns true when a hidden session_stop continuation turn was scheduled. */ async #emitSessionStopEvent( messages: AgentMessage[], lastAssistantMessage = this.getLastAssistantMessage(), - ): Promise { - if (this.#agentKind === "sub" || !this.#extensionRunner?.hasHandlers("session_stop")) return; + ): Promise { + if (this.#abortInProgress || this.#isDisposed) { + this.#resetSessionStopContinuationState(); + return false; + } + if (this.#agentKind === "sub" || !this.#extensionRunner?.hasHandlers("session_stop")) { + return false; + } const generation = this.#promptGeneration; const result = await this.#extensionRunner.emitSessionStop({ messages, @@ -5890,12 +6494,12 @@ export class AgentSession { }); if (this.#promptGeneration !== generation || this.#abortInProgress || this.#isDisposed) { this.#resetSessionStopContinuationState(); - return; + return false; } const additionalContext = this.#sessionStopContinuationContext(result); if (!additionalContext) { this.#resetSessionStopContinuationState(); - return; + return false; } if (this.#sessionStopContinuationCount >= SESSION_STOP_CONTINUATION_CAP) { logger.warn("session_stop continuation cap reached", { @@ -5903,7 +6507,7 @@ export class AgentSession { cap: SESSION_STOP_CONTINUATION_CAP, }); this.#resetSessionStopContinuationState(); - return; + return false; } this.#sessionStopContinuationCount++; this.#sessionStopHookActive = true; @@ -5918,6 +6522,7 @@ export class AgentSession { }, true, ); + return true; } /** Emit extension events based on session events */ @@ -6190,6 +6795,43 @@ export class AgentSession { await this.refreshBaseSystemPrompt(); } } + /** Run one abortable auto-learn capture outside the primary agent loop. */ + async runAutolearnCapture(capture: (signal: AbortSignal) => Promise): Promise { + if (this.#autolearnCaptureTask || this.#isDisposed) return; + const controller = new AbortController(); + this.#autolearnCaptureAbortController = controller; + const task = (async () => { + try { + await capture(controller.signal); + } catch (error) { + if (!controller.signal.aborted) throw error; + } finally { + if (this.#autolearnCaptureAbortController === controller) { + this.#autolearnCaptureAbortController = undefined; + } + } + })(); + this.#autolearnCaptureTask = task; + try { + await task; + } finally { + if (this.#autolearnCaptureTask === task) this.#autolearnCaptureTask = undefined; + } + } + + #abortAutolearnCapture(): void { + this.#autolearnCaptureAbortController?.abort(); + } + + async #drainAutolearnCapture(): Promise { + const task = this.#autolearnCaptureTask; + if (!task) return; + try { + await withTimeout(task, 3_000, "Timed out draining auto-learn capture during dispose"); + } catch (error) { + logger.warn("Auto-learn capture did not settle during dispose", { error: String(error) }); + } + } /** True once dispose() has begun; deferred background work (e.g. the deferred * MCP discovery task in sdk.ts) must not touch the session past this point. */ @@ -6212,6 +6854,8 @@ export class AgentSession { */ beginDispose(): void { this.#isDisposed = true; + this.#titleGenerationAbortController.abort(); + this.#abortAutolearnCapture(); this.#flushPendingIrcAsides(); this.yieldQueue.clear(); this.agent.setAsideMessageProvider(undefined); @@ -6236,129 +6880,136 @@ export class AgentSession { return this.#disposeCall; } + async #disposeOwnedAsyncJobs(): Promise { + this.#cancelOwnAsyncJobs(); + const manager = this.#ownedAsyncJobManager; + if (!manager) return; + + try { + const drained = await manager.dispose({ timeoutMs: 3_000 }); + const deliveryState = manager.getDeliveryState(); + if (drained === false && deliveryState) { + logger.warn("Async job completion deliveries still pending during dispose", { ...deliveryState }); + } + } finally { + if (AsyncJobManager.instance() === manager) { + AsyncJobManager.setInstance(undefined); + } + } + } + + async #disposeEvalKernels(): Promise { + const settled = await this.#prepareEvalExecutionsForDispose(); + if (!settled) { + logger.warn("Detaching retained eval-kernel ownership during dispose while eval execution is still active"); + } + + const results = await Promise.allSettled([ + disposeKernelSessionsByOwner(this.#evalKernelOwnerId), + disposeRubyKernelSessionsByOwner(this.#evalKernelOwnerId), + disposeJuliaKernelSessionsByOwner(this.#evalKernelOwnerId), + ]); + const errors: unknown[] = []; + for (const result of results) { + if (result.status === "rejected") errors.push(result.reason); + } + if (errors.length > 0) throw new AggregateError(errors, "Failed to dispose one or more eval kernels"); + } + + async #releaseOwnedBrowserTabs(ownerId: string | undefined): Promise { + if (!ownerId) return; + try { + const released = await withTimeout( + releaseTabsForOwner(ownerId, { kill: true }), + 3_000, + "Timed out releasing owned browser tabs during dispose", + ); + if (released > 0) { + logger.debug("Released owned browser tabs during dispose", { ownerId, released }); + } + } catch (error) { + logger.warn("Failed to release owned browser tabs during dispose", { error: String(error) }); + } + } + + async #disconnectOwnedMcp(): Promise { + if (!this.#disconnectOwnedMcpManager) return; + try { + await withTimeout( + this.#disconnectOwnedMcpManager(), + 3_000, + "Timed out disconnecting owned MCP manager during dispose", + ); + } catch (error) { + logger.warn("Failed to disconnect owned MCP manager during dispose", { error: String(error) }); + } + } + + async #disposeMnemopi( + state: MnemopiSessionState | undefined, + consolidateTimeoutMs: number | undefined, + ): Promise { + try { + await state?.dispose({ timeoutMs: consolidateTimeoutMs }); + } finally { + // Consolidation may embed final memories, so terminate its worker only afterward. + await shutdownMnemopiEmbedClient(); + } + } + async #doDispose(options: AgentSessionDisposeOptions = {}): Promise { this.beginDispose(); this.#recordSessionExit(options.reason ?? "dispose"); this.#cancelExitRecorder?.(); this.#cancelExitRecorder = undefined; try { - if (this.#extensionRunner?.hasHandlers("session_shutdown")) { - await this.#extensionRunner.emit({ type: "session_shutdown" }); - } + await emitSessionShutdownEvent(this.#extensionRunner); } catch (error) { logger.warn("Failed to emit session_shutdown event", { error: String(error) }); } - // Abort post-prompt work so the drain below can complete. Without this, a - // deferred-handoff task that has already advanced into - // `await this.handoff(...) → generateHandoff(...)` keeps awaiting a live LLM stream - // — Promise.allSettled() in #cancelPostPromptTasks then waits forever, freezing - // /exit and Ctrl+C-double-tap. The post-prompt task's own AbortSignal does not - // propagate into the inner handoff/compaction controllers, so we abort them - // explicitly. agent.abort() is needed for an agent.continue() that may have - // raced the deferred handoff (its streaming loop is awaited by the wrapper IIFE). - // - // Tool work (bash/eval/python) is NOT aborted here — those have their own - // dispose paths and shared kernels are contractually allowed to survive a - // session's dispose. + + // Stop fallback extension timers before aborting deferred work they could enqueue. + this.#fallbackExtensionTimers?.clearAll(); this.abortRetry(); this.abortCompaction(); const postPromptDrain = this.#cancelPostPromptTasks(); this.agent.abort(); - await postPromptDrain; - // Cancel jobs this agent registered so a subagent's teardown doesn't - // leak its background bash/task work into the parent's manager. Only - // the session that owns the manager goes on to dispose it (which itself - // nukes any leftover jobs and pending deliveries). - this.#cancelOwnAsyncJobs(); - const ownedAsyncManager = this.#ownedAsyncJobManager; - if (ownedAsyncManager) { - const drained = await ownedAsyncManager.dispose({ timeoutMs: 3_000 }); - const deliveryState = ownedAsyncManager.getDeliveryState(); - if (drained === false && deliveryState) { - logger.warn("Async job completion deliveries still pending during dispose", { ...deliveryState }); - } - if (AsyncJobManager.instance() === ownedAsyncManager) { - AsyncJobManager.setInstance(undefined); + try { + await withTimeout(postPromptDrain, 5_000, "Timed out draining post-prompt tasks during dispose"); + } catch (error) { + logger.warn("Post-prompt tasks still draining at dispose deadline", { error: String(error) }); + } + await this.#drainAutolearnCapture(); + + const hindsightState = this.getHindsightSessionState(); + const mnemopiState = setMnemopiSessionState(this, undefined); + const advisorRecorderClosed = this.#advisorRecorderClosed; + const results = await Promise.allSettled([ + this.#disposeOwnedAsyncJobs(), + this.#disposeEvalKernels(), + this.#releaseOwnedBrowserTabs(this.sessionManager.getSessionId()), + shutdownTinyTitleClient(), + this.#disconnectOwnedMcp(), + advisorRecorderClosed, + hindsightState?.flushRetainQueue() ?? Promise.resolve(), + this.#disposeMnemopi(mnemopiState, options.mnemopiConsolidateTimeoutMs), + ]); + for (const result of results) { + if (result.status === "rejected") { + logger.warn("Session dispose subsystem failed during parallel teardown", { + error: String(result.reason), + }); } } - const evalExecutionsSettled = await this.#prepareEvalExecutionsForDispose(); - if (!evalExecutionsSettled) { - logger.warn("Detaching retained eval-kernel ownership during dispose while eval execution is still active"); - } - await disposeKernelSessionsByOwner(this.#evalKernelOwnerId); - await disposeRubyKernelSessionsByOwner(this.#evalKernelOwnerId); - await disposeJuliaKernelSessionsByOwner(this.#evalKernelOwnerId); - // Release headless / spawned Chromium and worker tabs this session - // opened via the browser tool. The tool's `tabs`/`browsers` maps are - // module-global — subagents and future sessions share them — so we - // walk by `ownerSessionId` (assigned at `acquireTab` creation, never on - // reuse) and touch only what THIS session created. Bounded so a broken - // CDP close cannot stall `/exit`; mirrors the async-job/MCP pattern. - // (Issue #3963.) - const browserOwnerId = this.sessionManager.getSessionId(); - if (browserOwnerId) { - try { - const released = await withTimeout( - releaseTabsForOwner(browserOwnerId, { kill: true }), - 3_000, - "Timed out releasing owned browser tabs during dispose", - ); - if (released > 0) { - logger.debug("Released owned browser tabs during dispose", { ownerId: browserOwnerId, released }); - } - } catch (error) { - logger.warn("Failed to release owned browser tabs during dispose", { error: String(error) }); - } - } - await shutdownTinyTitleClient(); + this.#releasePowerAssertion(); - // Clean up an empty session created by this session's /move so it doesn't accumulate. await cleanupEmptyMoveSession(this.sessionManager, this.#movedFromEmptySessionFile); this.#movedFromEmptySessionFile = undefined; + // All teardown branches that can append session entries have settled. await this.sessionManager.close(); - // beginDispose() stopped the advisor and captured its recorder close; await - // it so the final advisor turn is flushed before the process may exit. - await this.#advisorRecorderClosed; this.#closeAllProviderSessions("dispose"); - // Disconnect the MCP manager this session OWNS so its stdio servers are - // not orphaned at exit. Best-effort: a failure here must never throw out - // of dispose. Only owning (top-level) sessions provide this callback; - // subagents reuse a parent's manager and must not tear it down. Idempotent - // with the deferred-discovery disconnect in `createAgentSession`. - // - // BOUNDED: an owned manager may hold an HTTP/SSE server whose session- - // termination DELETE blocks up to the MCP request timeout (30s default, - // unbounded when OMP_MCP_TIMEOUT_MS=0), so awaiting `disconnectAll()` - // unbounded would stall /exit and print-mode shutdown on a broken remote - // endpoint. Race it against a short deadline — stdio close (the subprocess - // reap this targets) completes well within the bound; a slow transport - // close is left to finish detached. Mirrors the bounded async-job teardown. - if (this.#disconnectOwnedMcpManager) { - try { - await withTimeout( - this.#disconnectOwnedMcpManager(), - 3_000, - "Timed out disconnecting owned MCP manager during dispose", - ); - } catch (error) { - logger.warn("Failed to disconnect owned MCP manager during dispose", { error: String(error) }); - } - } - // Flush the retain queue BEFORE clearing the session's pointer so - // `HindsightRetainQueue.#doFlush` still sees `session.getHindsightSessionState() === state`. - // Reversed, the spliced batch survives just long enough to fail the - // identity check and get dropped with a `session vanished` warning. - const hindsightState = this.getHindsightSessionState(); - await hindsightState?.flushRetainQueue(); this.setHindsightSessionState(undefined); hindsightState?.dispose(); - const mnemopiState = setMnemopiSessionState(this, undefined); - await mnemopiState?.dispose({ timeoutMs: options.mnemopiConsolidateTimeoutMs }); - // Tear down the embeddings subprocess AFTER mnemopi state.dispose: - // consolidate-on-dispose may still call `embed()` to store the final - // memories, and that round-trips through the worker we are about to - // hard-kill (issue #3031). - await shutdownMnemopiEmbedClient(); this.#disconnectFromAgent(); if (this.#unsubscribeAppendOnly) { this.#unsubscribeAppendOnly(); @@ -6418,6 +7069,12 @@ export class AgentSession { return this.agent.state.model; } + /** Resolved selector while retry routing is using a fallback model. */ + get retryFallbackModel(): string | undefined { + const model = this.model; + return this.#activeRetryFallback && model ? formatRetryFallbackSelector(model, this.thinkingLevel) : undefined; + } + /** Effective thinking level applied to the agent (the resolved level when `auto`). */ get thinkingLevel(): ThinkingLevel | undefined { return this.#thinkingLevel; @@ -6459,6 +7116,53 @@ export class AgentSession { await this.agent.waitForIdle(); await this.#waitForPostPromptRecovery(); } + /** + * Prevent advisor notes from starting hidden primary turns while a headless + * caller prints and drains the final primary response. + */ + prepareForHeadlessAdvisorDrain(): void { + this.#preserveAdvisorAdvice = true; + } + + async #waitForPendingAdvisorCardEvents(timeoutMs: number): Promise { + const deadline = Date.now() + Math.max(0, timeoutMs); + while (this.#pendingAdvisorCardEvents.size > 0) { + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) return false; + const settled = Promise.allSettled([...this.#pendingAdvisorCardEvents]).then(() => true as const); + const { promise: timedOut, resolve } = Promise.withResolvers(); + const timer = setTimeout(() => resolve(false), remainingMs); + try { + if (!(await Promise.race([settled, timedOut]))) return false; + } finally { + clearTimeout(timer); + } + } + return true; + } + + /** + * Wait for active advisor reviews and their emitted card events before a + * headless caller disposes the session. Returns `false` and logs work disposal + * will abandon when the shared deadline expires or an advisor fails. + */ + async waitForAdvisorCatchup(timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs; + const results = await Promise.all(this.#advisors.map(advisor => advisor.runtime.waitForCatchup(timeoutMs, 1))); + const cardEventsCaughtUp = await this.#waitForPendingAdvisorCardEvents(Math.max(0, deadline - Date.now())); + const abandoned = this.#advisors.filter( + (advisor, index) => results[index] === false && advisor.runtime.backlog > 0, + ); + if (abandoned.length > 0 || !cardEventsCaughtUp) { + logger.warn("advisor shutdown drain incomplete; disposal will abandon reviews or cards", { + timeoutMs, + advisors: abandoned.map(advisor => ({ name: advisor.name, backlog: advisor.runtime.backlog })), + pendingAdvisorCards: this.#pendingAdvisorCardEvents.size, + }); + return false; + } + return true; + } async drainAsyncJobDeliveriesForAcp(options?: { timeoutMs?: number }): Promise { const manager = this.#asyncJobManager; @@ -6477,9 +7181,14 @@ export class AgentSession { } } - /** Most recent assistant message in agent state. */ + /** + * Most recent settled assistant message. A classifier-refusal turn pruned + * from active context at settle is still reported until the next run + * starts, so terminal-outcome consumers (print mode, task executor) see + * the refusal error rather than the previous turn — or nothing. + */ getLastAssistantMessage(): AssistantMessage | undefined { - return this.#findLastAssistantMessage(); + return this.#prunedTerminalRefusal ?? this.#findLastAssistantMessage(); } /** Current effective system prompt blocks (includes any per-turn extension modifications) */ get systemPrompt(): string[] { @@ -6739,82 +7448,187 @@ export class AgentSession { async #applyActiveToolsByName(toolNames: string[]): Promise { toolNames = normalizeToolNames(toolNames); + const selectedTools = toolNames.flatMap(name => { + const tool = this.#toolRegistry.get(name); + return tool ? [{ name, tool }] : []; + }); + const xdevReadAvailable = this.#builtInToolNames.has("read") && selectedTools.some(({ name }) => name === "read"); + const isPresentationPinned = (name: string): boolean => + this.#presentationPinnedToolNames?.has(name) === true || this.#runtimeSelectedToolNames?.has(name) === true; + const mountCandidates = selectedTools.filter( + ({ name, tool }) => + this.#xdevRegistry !== undefined && + xdevReadAvailable && + !isPresentationPinned(name) && + isMountableUnderXdev(tool), + ); + + let builtInWriteAvailable = this.#builtInToolNames.has("write"); + if (mountCandidates.length > 0 && !builtInWriteAvailable) { + builtInWriteAvailable = (await this.#ensureWriteRegistered?.()) === true; + if (builtInWriteAvailable) this.#builtInToolNames.add("write"); + } + const mountNames = builtInWriteAvailable ? new Set(mountCandidates.map(({ name }) => name)) : new Set(); const tools: AgentTool[] = []; const validToolNames: string[] = []; const mountedTools: AgentTool[] = []; - for (const name of toolNames) { - const tool = this.#toolRegistry.get(name); - if (!tool) continue; - // Discoverable tools are presented as `xd://` devices (kept out of the - // top-level schema) when the transport is active; everything else stays - // top-level. `loadMode` decides presentation only — selection is upstream. - if (this.#xdevRegistry && isMountableUnderXdev(tool)) { - mountedTools.push(tool); + for (const { name, tool } of selectedTools) { + if (mountNames.has(name)) { + mountedTools.push(this.#wrapToolForAcpPermission(tool)); } else { tools.push(this.#wrapToolForAcpPermission(tool)); validToolNames.push(name); } } - // Reconcile the dynamic `xd://` mounts: newly-active discoverable tools are - // mounted, deactivated ones dropped (built-in devices are preserved). A - // removed or disconnected tool must not stay callable through a stale device. + + const pinnedWrite = isPresentationPinned("write"); + const activeDeferrableTool = tools.some(tool => tool.deferrable === true); + const transportNeeded = mountedTools.length > 0 || activeDeferrableTool || this.#planModeState?.enabled === true; + if (transportNeeded && !builtInWriteAvailable) { + builtInWriteAvailable = (await this.#ensureWriteRegistered?.()) === true; + if (builtInWriteAvailable) this.#builtInToolNames.add("write"); + } + if (transportNeeded && builtInWriteAvailable) { + const write = this.#toolRegistry.get("write"); + if (write && !validToolNames.includes("write")) { + tools.push(this.#wrapToolForAcpPermission(write)); + validToolNames.push("write"); + } + } else if ( + !pinnedWrite && + (this.#presentationPinnedToolNames !== undefined || this.#runtimeSelectedToolNames !== undefined) + ) { + const writeNameIndex = validToolNames.indexOf("write"); + if (writeNameIndex >= 0 && this.#builtInToolNames.has("write")) validToolNames.splice(writeNameIndex, 1); + const writeToolIndex = tools.findIndex(tool => tool.name === "write" && this.#builtInToolNames.has("write")); + if (writeToolIndex >= 0) tools.splice(writeToolIndex, 1); + } + const previousMounted = this.#mountedXdevToolNames; + const previousMountedTools = [...previousMounted].flatMap(name => { + const tool = this.#xdevRegistry?.get(name); + return tool ? [tool] : []; + }); + const previousActiveToolNames = this.getActiveToolNames(); this.#mountedXdevToolNames = new Set(mountedTools.map(tool => tool.name)); this.#xdevRegistry?.reconcile(mountedTools); - this.#notifyXdevMountDelta(previousMounted); this.#setActiveToolNames?.(validToolNames); - this.agent.setTools(tools); - // Rebuild base system prompt with new tool set, but only when the tool set - // actually changed. MCP servers can reconnect at arbitrary times and call - // `refreshMCPTools` -> `#applyActiveToolsByName` even though the resulting - // tool list is byte-identical. Skipping the rebuild keeps the system prompt - // stable, which is required for Anthropic prompt caching to keep hitting. - if (this.#rebuildSystemPrompt) { - const signature = this.#computeAppliedToolSignature(validToolNames, tools); - if (signature !== this.#lastAppliedToolSignature) { - if (this.#lastAppliedToolSignature !== undefined) { - this.#clearInheritedProviderPromptCacheKey(); + let rebuiltSystemPrompt: string[] | undefined; + let rebuiltSignature: string | undefined; + try { + if (this.#rebuildSystemPrompt) { + const signature = this.#computeAppliedToolSignature(validToolNames, tools); + if (signature !== this.#lastAppliedToolSignature) { + const built = await this.#rebuildSystemPrompt(validToolNames, this.#toolRegistry); + rebuiltSystemPrompt = built.systemPrompt; + rebuiltSignature = signature; } - const built = await this.#rebuildSystemPrompt(validToolNames, this.#toolRegistry); - this.#baseSystemPrompt = built.systemPrompt; - this.#baseSystemPromptBeforeMemoryPromotion = undefined; - this.agent.setSystemPrompt(this.#baseSystemPrompt); - this.#lastAppliedToolSignature = signature; - this.#promptModelKey = this.#currentPromptModelKey(); } + } catch (error) { + this.#mountedXdevToolNames = previousMounted; + this.#xdevRegistry?.reconcile(previousMountedTools); + this.#setActiveToolNames?.(previousActiveToolNames); + throw error; + } + + this.#notifyXdevMountDelta(previousMounted); + this.agent.setTools(tools); + if (rebuiltSystemPrompt && rebuiltSignature) { + if (this.#lastAppliedToolSignature !== undefined) this.#clearInheritedProviderPromptCacheKey(); + this.#baseSystemPrompt = rebuiltSystemPrompt; + this.#baseSystemPromptBeforeMemoryPromotion = undefined; + this.agent.setSystemPrompt(this.#baseSystemPrompt); + this.#lastAppliedToolSignature = rebuiltSignature; + this.#promptModelKey = this.#currentPromptModelKey(); } } /** - * Announce a mid-session `xd://` mount delta to the model as a steered - * system notice instead of rewriting the system prompt: the prompt (and - * its provider cache prefix) stays byte-stable across MCP connects and - * disconnects, and the model learns about new devices from the notice - * (docs + schema stay one `read xd://` away). The full docs join - * the system prompt opportunistically on the next unrelated rebuild. + * Record a mid-session `xd://` mount delta for the model without rewriting + * the system prompt: the prompt (and its provider cache prefix) stays + * byte-stable across MCP connects and disconnects. The delta is NOT steered + * immediately — a steered notice landing at a run's stop boundary (or while + * the session is idle) forces an unsolicited extra assistant turn — it is + * coalesced into {@link #pendingXdevMountDelta} and rides along with the + * next prompt (docs + schema stay one `read xd://` away). The full + * docs join the system prompt opportunistically on the next unrelated + * rebuild. */ #notifyXdevMountDelta(previousMounted: ReadonlySet): void { const registry = this.#xdevRegistry; if (!registry) return; const current = this.#mountedXdevToolNames; const addedNames = [...current].filter(name => !previousMounted.has(name)); - const removed = [...previousMounted].filter(name => !current.has(name)).map(name => ({ name })); - if (addedNames.length === 0 && removed.length === 0) return; - const summaries = new Map(registry.entries().map(entry => [entry.name, entry.summary])); - const added = addedNames.map(name => ({ name, summary: summaries.get(name) ?? "" })); - this.agent.steer({ + const removedNames = [...previousMounted].filter(name => !current.has(name)); + if (addedNames.length === 0 && removedNames.length === 0) return; + // Coalesce against the unannounced delta: an unmount cancels a pending + // mount the model never learned about, and a remount cancels a pending + // unmount. + const pending = this.#pendingXdevMountDelta ?? { added: new Set(), removed: new Set() }; + for (const name of addedNames) { + if (!pending.removed.delete(name)) pending.added.add(name); + } + for (const name of removedNames) { + if (!pending.added.delete(name)) pending.removed.add(name); + } + this.#pendingXdevMountDelta = pending.added.size > 0 || pending.removed.size > 0 ? pending : undefined; + if (this.settings.get("startup.quiet")) return; + const parts: string[] = []; + if (addedNames.length > 0) parts.push(`mounted ${addedNames.join(", ")}`); + if (removedNames.length > 0) parts.push(`unmounted ${removedNames.join(", ")}`); + this.emitNotice("info", `xd://: ${parts.join("; ")}`, "xdev"); + } + + /** + * Render and consume the pending xd:// mount delta as a hidden notice, or + * `undefined` when nothing unannounced is queued. Called from the prompt + * paths so the notice rides along with user input instead of forcing its + * own model turn. + */ + #takePendingXdevMountNotice(): CustomMessage | undefined { + const pending = this.#pendingXdevMountDelta; + if (!pending) return undefined; + this.#pendingXdevMountDelta = undefined; + const summaries = new Map(this.#xdevRegistry?.entries().map(entry => [entry.name, entry.summary]) ?? []); + const added = [...pending.added].map(name => ({ name, summary: summaries.get(name) ?? "" })); + const removed = [...pending.removed].map(name => ({ name })); + return { role: "custom", customType: XDEV_MOUNT_NOTICE_MESSAGE_TYPE, content: prompt.render(xdevMountNoticePrompt, { added, removed }), attribution: "agent", display: false, timestamp: Date.now(), + }; + } + + /** + * Rediscover disk-backed skills and rebuild prompt-facing state without + * recreating the session. Explicit skill snapshots (`--no-skills`, + * SDK-provided `skills`) remain fixed for the lifetime of the session. + */ + async refreshSkills(): Promise { + if (!this.#skillsReloadable) { + return; + } + + resetCapabilities(); + const skillsSettings = this.settings.getGroup("skills"); + const discovered = await loadSkills({ + ...skillsSettings, + cwd: this.sessionManager.getCwd(), + disabledExtensions: this.settings.get("disabledExtensions") ?? [], }); - const parts: string[] = []; - if (added.length > 0) parts.push(`mounted ${added.map(entry => entry.name).join(", ")}`); - if (removed.length > 0) parts.push(`unmounted ${removed.map(entry => entry.name).join(", ")}`); - this.emitNotice("info", `xd://: ${parts.join("; ")}`, "xdev"); + this.#skills = discovered.skills; + this.#skillWarnings = discovered.warnings; + this.#skillsSettings = skillsSettings; + + if (this.#agentKind === "main") { + setActiveSkills(this.#skills); + } + await this.refreshBaseSystemPrompt(); + this.#notifyCommandMetadataChanged(); } /** @@ -6824,7 +7638,67 @@ export class AgentSession { * Changes take effect before the next model call. */ async setActiveToolsByName(toolNames: string[]): Promise { - await this.#applyActiveToolsByName(toolNames); + const normalized = normalizeToolNames(toolNames); + // Transport-write eligibility keys off the *current* active set: an ordinary + // selection change should not demote `write` unless it is already active. + await this.#applyToolPresentation( + normalized, + this.#mountedXdevToolNames, + this.getActiveToolNames().includes("write"), + ); + } + + /** + * Restore an enabled tool set with its exact top-level versus `xd://` partition. + * + * Both inputs are required because {@link setActiveToolsByName} only receives the + * enabled name list and classifies mounts from the *current* `#mountedXdevToolNames`. + * Rollback/restore callers must pass the snapshotted mounted subset so names that + * were top-level stay pinned (`#runtimeSelectedToolNames`) and names that were under + * `xd://` remain mount-eligible, even when the live mount set has drifted. + * + * Names outside `mountedToolNames` are pinned top-level for this application; + * names in the mounted subset remain eligible for xdev mounting. Delegates the + * actual apply through `#applyActiveToolsByName` and restores the prior runtime + * selection if that apply throws. + */ + async setActiveToolPresentation(toolNames: string[], mountedToolNames: string[]): Promise { + const normalized = normalizeToolNames(toolNames); + // Restoration targets a snapshot, so write eligibility comes from the + // *target* set rather than whatever happens to be active mid-rollback. + await this.#applyToolPresentation( + normalized, + new Set(normalizeToolNames(mountedToolNames)), + normalized.includes("write"), + ); + } + + /** + * Shared body for {@link setActiveToolsByName} and {@link setActiveToolPresentation}: + * pins non-mounted names as the runtime selection (holding `write` back when it is + * transport-only) and applies the set, rolling the selection back if apply throws. + */ + async #applyToolPresentation( + normalized: string[], + mounted: ReadonlySet, + writeSelected: boolean, + ): Promise { + const transportWriteActive = + writeSelected && + this.#builtInToolNames.has("write") && + this.#presentationPinnedToolNames?.has("write") !== true && + this.#runtimeSelectedToolNames?.has("write") !== true && + (mounted.size > 0 || this.#planModeState?.enabled === true); + const previousRuntimeSelectedToolNames = this.#runtimeSelectedToolNames; + this.#runtimeSelectedToolNames = new Set( + normalized.filter(name => !mounted.has(name) && !(name === "write" && transportWriteActive)), + ); + try { + await this.#applyActiveToolsByName(normalized); + } catch (error) { + this.#runtimeSelectedToolNames = previousRuntimeSelectedToolNames; + throw error; + } } /** Rebuild the base system prompt using the current active tool set. */ @@ -6967,6 +7841,12 @@ export class AgentSession { */ async refreshMCPTools(mcpTools: CustomTool[]): Promise { const existingNames = Array.from(this.#toolRegistry.keys()); + const previousMcpTools = new Map( + existingNames.flatMap(name => { + const tool = this.#toolRegistry.get(name); + return isMCPToolName(name) && tool ? [[name, tool] as const] : []; + }), + ); for (const name of existingNames) { if (isMCPToolName(name)) { this.#toolRegistry.delete(name); @@ -6994,10 +7874,18 @@ export class AgentSession { this.#toolRegistry.set(finalTool.name, finalTool); } - // Every connected MCP tool is enabled; re-derive the active set from the - // current non-MCP tools plus all freshly registered MCP tools. + // Every connected MCP tool is selected; centralized repartitioning owns + // presentation pins and write-transport activation/removal. const nextActive = [...new Set([...this.#getActiveNonMCPToolNames(), ...mcpTools.map(tool => tool.name)])]; - await this.#applyActiveToolsByName(nextActive); + try { + await this.#applyActiveToolsByName(nextActive); + } catch (error) { + for (const name of this.#toolRegistry.keys()) { + if (isMCPToolName(name)) this.#toolRegistry.delete(name); + } + for (const [name, tool] of previousMcpTools) this.#toolRegistry.set(name, tool); + throw error; + } } /** @@ -7018,6 +7906,12 @@ export class AgentSession { const previousRpcHostToolNames = new Set(this.#rpcHostToolNames); const previousActiveToolNames = this.getEnabledToolNames(); + const previousRpcHostTools = new Map( + [...previousRpcHostToolNames].flatMap(name => { + const tool = this.#toolRegistry.get(name); + return tool ? [[name, tool] as const] : []; + }), + ); for (const name of previousRpcHostToolNames) { this.#toolRegistry.delete(name); } @@ -7039,9 +7933,16 @@ export class AgentSession { const autoActivatedRpcToolNames = rpcTools .filter(tool => !tool.hidden && !previousRpcHostToolNames.has(tool.name)) .map(tool => tool.name); - await this.#applyActiveToolsByName( - Array.from(new Set([...activeNonRpcToolNames, ...preservedRpcToolNames, ...autoActivatedRpcToolNames])), - ); + try { + await this.#applyActiveToolsByName( + Array.from(new Set([...activeNonRpcToolNames, ...preservedRpcToolNames, ...autoActivatedRpcToolNames])), + ); + } catch (error) { + for (const name of this.#rpcHostToolNames) this.#toolRegistry.delete(name); + this.#rpcHostToolNames = previousRpcHostToolNames; + for (const [name, tool] of previousRpcHostTools) this.#toolRegistry.set(name, tool); + throw error; + } } /** Whether auto-compaction is currently running */ @@ -7061,6 +7962,12 @@ export class AgentSession { return this.#postPromptTasks.size > 0; } + /** Register post-prompt work in tests without driving a full agent turn. */ + trackPostPromptTaskForTests(task: Promise): void { + if (!isBunTestRuntime()) throw new Error("trackPostPromptTaskForTests is test-only"); + this.#trackPostPromptTask(task); + } + /** All messages including custom types like BashExecutionMessage */ get messages(): AgentMessage[] { return this.agent.state.messages; @@ -7369,6 +8276,11 @@ export class AgentSession { .filter((tool): tool is AgentTool => tool !== undefined) .map(tool => this.#wrapToolForAcpPermission(tool)); this.agent.setTools(activeTools); + const mountedTools = [...this.#mountedXdevToolNames] + .map(name => this.#toolRegistry.get(name)) + .filter((tool): tool is AgentTool => tool !== undefined) + .map(tool => this.#wrapToolForAcpPermission(tool)); + this.#xdevRegistry?.reconcile(mountedTools); } #clearCheckpointRuntimeState(): void { @@ -7802,19 +8714,18 @@ export class AgentSession { timestamp, }); } - if ( - this.#magicKeywordEnabled("workflow") && - containsWorkflow(text) && - this.getActiveToolNames().includes("task") - ) { - keywordNotices.push({ - role: "custom", - customType: "workflow-notice", - content: renderWorkflowNotice({ taskBatch: this.settings.get("task.batch") }), - display: false, - attribution: "user", - timestamp, - }); + if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text)) { + const activeToolNames = this.getActiveToolNames(); + if (activeToolNames.includes("task") && activeToolNames.includes("eval")) { + keywordNotices.push({ + role: "custom", + customType: "workflow-notice", + content: renderWorkflowNotice({ taskBatch: this.settings.get("task.batch") }), + display: false, + attribution: "user", + timestamp, + }); + } } return keywordNotices; } @@ -8030,7 +8941,7 @@ export class AgentSession { const generation = this.#promptGeneration; try { // Flush any pending bash messages before the new prompt - this.#flushPendingBashMessages(); + await this.#flushPendingBashMessages(); this.#flushPendingPythonMessages(); this.#flushPendingIrcAsides(); @@ -8062,12 +8973,16 @@ export class AgentSession { ); } - // Check if we need to compact before sending (catches aborted responses). Run - // inline (allowDefer=false) so the handoff/maintenance fully settles before this - // prompt's agent loop starts — otherwise a deferred handoff would fire on the - // next microtask alongside the new turn. + // Recover a previously failed/incomplete assistant turn before sending. + // Successful historical turns take the cheaper pre-prompt threshold path + // below; re-running the full post-turn check on resume can synchronously + // rewrite/re-render old context before the new prompt starts. const lastAssistant = this.#findLastAssistantMessage(); - if (lastAssistant && !options?.skipCompactionCheck) { + if ( + lastAssistant && + !options?.skipCompactionCheck && + (lastAssistant.stopReason === "error" || lastAssistant.stopReason === "length") + ) { await this.#checkCompaction(lastAssistant, false, false, false); } @@ -8095,14 +9010,19 @@ export class AgentSession { messages.push(...options.prependMessages); } - messages.push(message); - // Early bail-out: if a newer abort/prompt cycle started during setup, // return before mutating shared state (nextTurn messages, system prompt). if (this.#promptGeneration !== generation) { return; } + // A pending xd:// delta accompanies the next user-authored prompt, + // never an agent-initiated continuation. + const xdevMountNotice = isUserQueuedMessage(message) ? this.#takePendingXdevMountNotice() : undefined; + if (xdevMountNotice) { + messages.push(xdevMountNotice); + } + messages.push(message); // Inject any pending "nextTurn" messages as context alongside the user message for (const msg of this.#pendingNextTurnMessages) { messages.push(msg); @@ -8264,8 +9184,11 @@ export class AgentSession { }, hasPendingMessages: () => this.queuedMessageCount > 0, shutdown: () => { - void this.dispose(); - process.exit(0); + // Await the idempotent dispose() before exiting so the browser + // reaper and other bounded teardown complete — a fire-and-forget + // `void this.dispose()` raced process.exit() and could leave an + // OMP-owned Chromium alive (#5643). + void this.dispose().finally(() => process.exit(0)); }, getContextUsage: () => this.getContextUsage(), waitForIdle: () => this.waitForIdle(), @@ -8301,9 +9224,20 @@ export class AgentSession { await this.reload(); }, getSystemPrompt: () => this.systemPrompt, + setInterval: (callback, ms, ...args) => this.#fallbackTimers().setInterval(callback, ms, ...args), + setTimeout: (callback, ms, ...args) => this.#fallbackTimers().setTimeout(callback, ms, ...args), + clearTimer: timer => this.#fallbackTimers().clear(timer), }; } + /** Lazily create the runner-less command-context timer registry (#5664). */ + #fallbackTimers(): ManagedTimers { + this.#fallbackExtensionTimers ??= new ManagedTimers((event, error) => + logger.warn("Extension timer callback threw", { event, error }), + ); + return this.#fallbackExtensionTimers; + } + /** * Try to execute a custom command. Returns the prompt string if found, null otherwise. * If the command returns void, returns empty string to indicate it was handled. @@ -8916,6 +9850,13 @@ export class AgentSession { } #scheduleReplanTitleRefresh(): void { + // Headless subagent sessions have no operator-visible title, so a todo-init + // replan refresh only burns a tiny-model call whose result lands in JSONL + // and is never shown (issue #5910). In an interactive host the operator can + // focus a live subagent from the Agent Hub, where the status line renders + // its session name — so keep the refresh there and only skip subagents when + // no focusable UI exists (print/RPC/ACP/eval/SDK/CI). + if (this.#agentKind === "sub" && !isInteractiveHost()) return; if (this.#replanTitleRefreshInFlight) return; if (!this.settings.get("title.refreshOnReplan")) return; if (this.sessionManager.titleSource === "user") return; @@ -8937,16 +9878,26 @@ export class AgentSession { this.#replanTitleRefreshInFlight = refresh; } - async #refreshTitleAfterReplan(context: string, sessionId: string): Promise { - const title = await generateSessionTitle( - context, + /** + * Generate an automatic session title tied to this session's lifecycle. + * Input and replan callers share the signal so disposal cancels provider and + * local-worker requests instead of leaving background inference alive. + */ + generateTitle(firstMessage: string): Promise { + return generateSessionTitle( + firstMessage, this.#modelRegistry, this.settings, - sessionId, + this.sessionId, this.model, provider => this.agent.metadataForProvider(provider), this.#titleSystemPrompt, + this.#titleGenerationAbortController.signal, ); + } + + async #refreshTitleAfterReplan(context: string, sessionId: string): Promise { + const title = await this.generateTitle(context); if (!title) return; if (this.sessionManager.getSessionId() !== sessionId) return; if (!this.settings.get("title.refreshOnReplan")) return; @@ -9014,6 +9965,7 @@ export class AgentSession { // auto-starting a fresh turn during cleanup. this.#abortInProgress = true; try { + this.#abortAutolearnCapture(); this.abortRetry(); this.#promptGeneration++; this.#scheduledHiddenNextTurnGeneration = undefined; @@ -9035,6 +9987,7 @@ export class AgentSession { this.agent.abort(options?.reason); await postPromptDrain; await this.agent.waitForIdle(); + await this.#drainAutolearnCapture(); await this.#goalRuntime.onTaskAborted({ reason: options?.goalReason ?? "interrupted" }); // Clear prompt-in-flight state: waitForIdle resolves when the agent loop's finally // block runs, but nested prompt setup/finalizers may still be unwinding. Without this, @@ -9094,27 +10047,36 @@ export class AgentSession { await this.abort(); this.#cancelOwnAsyncJobs(); this.#closeAllProviderSessions("new session"); - this.agent.reset(); - if (options?.drop && previousSessionFile) { - // Detach the advisor recorder feed and drain its writer BEFORE deleting the - // old artifacts dir: `await this.abort()` only stops the primary, so a still- - // running advisor turn could otherwise finish, emit `message_end`, and recreate - // `/__advisor.jsonl`. #resetAdvisorSessionState (after newSession) re-primes - // the advisor and re-attaches the feed at the new session's path. - for (const a of this.#advisors) { - a.agentUnsubscribe?.(); - a.agentUnsubscribe = undefined; - await a.recorder.close(); + await this.#flushPendingBashMessages(); + const bashTransition = this.#beginBashSessionTransition({ persistDetached: options?.drop !== true }); + let sessionTransitioned = false; + try { + this.agent.reset(); + if (options?.drop && previousSessionFile) { + // Detach the advisor recorder feed and drain its writer BEFORE deleting the + // old artifacts dir: `await this.abort()` only stops the primary, so a still- + // running advisor turn could otherwise finish, emit `message_end`, and recreate + // `/__advisor.jsonl`. #resetAdvisorSessionState (after newSession) re-primes + // the advisor and re-attaches the feed at the new session's path. + for (const a of this.#advisors) { + a.agentUnsubscribe?.(); + a.agentUnsubscribe = undefined; + await a.recorder.close(); + } + try { + await this.sessionManager.dropSession(previousSessionFile); + } catch (err) { + logger.error("Failed to delete session during /drop", { err }); + } + } else { + await this.sessionManager.flush(); } - try { - await this.sessionManager.dropSession(previousSessionFile); - } catch (err) { - logger.error("Failed to delete session during /drop", { err }); - } - } else { - await this.sessionManager.flush(); + await this.sessionManager.newSession(options); + this.#markBashSessionTransition(bashTransition); + sessionTransitioned = true; + } finally { + this.#finishBashSessionTransition(bashTransition, sessionTransitioned); } - await this.sessionManager.newSession(options); this.#clearCheckpointRuntimeState(); this.setTodoPhases([]); @@ -9180,14 +10142,25 @@ export class AgentSession { } } + await this.#flushPendingBashMessages(); // Flush current session to ensure all entries are written await this.sessionManager.flush(); + const bashTransition = this.#beginBashSessionTransition(); // Fork the session (creates new session file with same entries) - const forkResult = await this.sessionManager.fork(); + let forkResult: { oldSessionFile: string; newSessionFile: string } | undefined; + try { + forkResult = await this.sessionManager.fork(); + } catch (error) { + this.#finishBashSessionTransition(bashTransition, false); + throw error; + } if (!forkResult) { + this.#finishBashSessionTransition(bashTransition, false); return false; } + this.#markBashSessionTransition(bashTransition); + this.#finishBashSessionTransition(bashTransition, true); // Copy artifacts directory if it exists const oldArtifactDir = forkResult.oldSessionFile.slice(0, -6); @@ -9520,15 +10493,16 @@ export class AgentSession { } /** - * Set the thinking level. `auto` enables per-turn classification; the selector - * itself is never written to the session log, but resolved concrete levels are - * persisted when real user turns are classified so resumed sessions keep the - * last resolved effort instead of reverting to pending auto. + * Set the thinking level. `auto` enables per-turn classification. Entering + * auto writes its provisional level plus `configured: "auto"` immediately, + * giving external readers an authoritative selection receipt before the next + * user turn. Later classifications persist only changed concrete resolutions. */ setThinkingLevel(level: ConfiguredThinkingLevel | undefined, persist: boolean = false): void { if (level === AUTO_THINKING) { const provisional = resolveProvisionalAutoLevel(this.model); const wasAuto = this.#autoThinking; + const previousLevel = this.#thinkingLevel; this.#autoThinking = true; this.#autoResolvedLevel = undefined; this.#thinkingLevel = provisional; @@ -9539,7 +10513,9 @@ export class AgentSession { if (persist) { this.settings.set("defaultThinkingLevel", AUTO_THINKING); } - if (!wasAuto || this.#thinkingLevel !== provisional) { + const isChanging = !wasAuto || previousLevel !== provisional; + if (isChanging) { + this.sessionManager.appendThinkingLevelChange(provisional, AUTO_THINKING); this.#emit({ type: "thinking_level_changed", thinkingLevel: provisional, configured: AUTO_THINKING }); } return; @@ -9648,7 +10624,7 @@ export class AgentSession { const effort = resolved ?? resolveProvisionalAutoLevel(model); if (effort === undefined) return; - const shouldPersistResolution = this.#autoResolvedLevel !== effort; + const shouldPersistResolution = this.#thinkingLevel !== effort; this.#autoResolvedLevel = effort; this.#thinkingLevel = effort; this.#applyThinkingLevelToAgent(effort); @@ -10370,6 +11346,13 @@ export class AgentSession { this.#compactionAbortController = undefined; } this.#reconnectToAgent(); + // Compaction disconnected before `await abort()`, so abort's finally drain + // (and any steer/follow-up that arrived mid-compaction — async IRC, an + // `xd://` mount notice, an SDK/RPC steer) was suppressed while disconnected + // (issue #5800). Unlike `/new`/switchSession, compaction preserves the agent + // queues, so nothing else resumes them: re-drain now that the listener is back + // and `isCompacting` is false, or the queued turn hangs until the next prompt. + this.#drainStrandedQueuedMessages(); } } @@ -10563,9 +11546,20 @@ export class AgentSession { return undefined; } } + await this.#flushPendingBashMessages(); await this.sessionManager.flush(); + const bashTransition = this.#beginBashSessionTransition(); this.#cancelOwnAsyncJobs(); - await this.sessionManager.newSession(previousSessionFile ? { parentSession: previousSessionFile } : undefined); + let sessionTransitioned = false; + try { + await this.sessionManager.newSession( + previousSessionFile ? { parentSession: previousSessionFile } : undefined, + ); + this.#markBashSessionTransition(bashTransition); + sessionTransitioned = true; + } finally { + this.#finishBashSessionTransition(bashTransition, sessionTransitioned); + } this.#clearCheckpointRuntimeState(); // agent.reset() clears the core steering/follow-up queues. Preserve any queued @@ -11025,6 +12019,7 @@ export class AgentSession { autoContinue, triggerContextTokens: postMaintenanceContextTokens, phase: "pre_turn", + terminalTextAnswer: isTerminalTextAssistantAnswer(assistantMessage), }); } logger.debug("Auto-compaction threshold satisfied but context promotion took over", { @@ -11081,7 +12076,17 @@ export class AgentSession { this.#pendingRecoveredRetryErrors = []; } - async #persistRetryLifecycleErrorMessage(message: AssistantMessage): Promise { + /** + * Durably record a terminal empty error turn (`stopReason: "error"` with no + * substantive content) that `#persistSessionMessageIfMissing` skipped, so the + * session JSONL keeps a record of why the run stopped instead of ending at the + * last tool result. A no-op for non-empty/non-error turns and idempotent via + * the already-persisted guard; the turn is dropped from active context by the + * caller (or `isProviderRefusalMessage`/`isEmptyErrorTurn` filters) so it is + * never replayed on the wire. Used by the retry-lifecycle dead-ends and the + * non-retry terminal error tail. + */ + async #persistTerminalEmptyErrorTurn(message: AssistantMessage): Promise { await this.#waitForSessionMessagePersistence(message); if (!isEmptyErrorTurn(message)) return; if (this.#sessionMessageAlreadyPersisted(message)) return; @@ -11123,7 +12128,7 @@ export class AgentSession { id: number, options: { switchedCredential: boolean; switchedModel: boolean; delayMs: number }, ): Promise { - await this.#persistRetryLifecycleErrorMessage(message); + await this.#persistTerminalEmptyErrorTurn(message); const persistenceKey = sessionMessagePersistenceKey(message); if (!persistenceKey) return; let branchEntry: SessionEntry | undefined; @@ -11220,7 +12225,8 @@ export class AgentSession { this.#emptyStopRetryCount++; if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) { const attempts = this.#emptyStopRetryCount - 1; - const finalError = "Assistant returned empty stop after retry cap"; + const finalError = + "Assistant returned empty stop after retry cap; try switching models or `/shake images` to remove archived frames"; logger.warn(finalError, { attempts, model: assistantMessage.model, @@ -11235,11 +12241,11 @@ export class AgentSession { this.#clearPendingRecoveredRetryErrors(); this.#retryAttempt = 0; this.#resolveRetry(); - // Tool-use orphans corrupt Anthropic message history (tool_result without - // matching tool_use). Always remove them even when the retry cap is hit. - if (assistantMessage.stopReason === "toolUse") { - this.#discardAssistantTurn(assistantMessage); - } + // A zero-content turn carries no transcript value, while its provider usage + // can anchor the next prompt at the full failed-request size and re-trigger + // compaction at the same boundary. Remove every capped empty stop; toolUse + // orphans still need this for Anthropic message-history validity. + await this.#dropPersistedAssistantTurn(assistantMessage); return false; } this.#discardAssistantTurn(assistantMessage); @@ -11256,11 +12262,12 @@ export class AgentSession { #isEmptyAssistantStop(assistantMessage: AssistantMessage): boolean { switch (assistantMessage.stopReason) { case "stop": - // Reasoning/thinking-only turns are not actionable: they do not - // answer the user and do not give the agent loop a tool call to run. + // Unsigned thinking alone is not actionable, but a signature is + // provider-authenticated content and makes the stop terminal. for (const content of assistantMessage.content) { if (content.type === "toolCall") return false; if (content.type === "text" && hasNonWhitespace(content.text)) return false; + if (content.type === "thinking" && hasNonWhitespace(content.thinkingSignature ?? "")) return false; } return true; case "toolUse": @@ -11458,11 +12465,13 @@ export class AgentSession { if (!branchEntry) return; const targetParentId = prunePrompt ? parentEntry.parentId : branchEntry.parentId; - if (targetParentId === null) { - this.sessionManager.resetLeaf(); - } else { - this.sessionManager.branch(targetParentId); - } + this.#withBashBranchTransition(() => { + if (targetParentId === null) { + this.sessionManager.resetLeaf(); + } else { + this.sessionManager.branch(targetParentId); + } + }); this.sessionManager.appendCustomEntry("accepted-terminal-empty-stop"); } @@ -11489,11 +12498,13 @@ export class AgentSession { if (!branchEntry) { return; } - if (branchEntry.parentId === null) { - this.sessionManager.resetLeaf(); - } else { - this.sessionManager.branch(branchEntry.parentId); - } + this.#withBashBranchTransition(() => { + if (branchEntry.parentId === null) { + this.sessionManager.resetLeaf(); + } else { + this.sessionManager.branch(branchEntry.parentId); + } + }); } #isSameAssistantMessage(left: AssistantMessage, right: AssistantMessage): boolean { @@ -11530,8 +12541,10 @@ export class AgentSession { if (this.#pendingRewindReport) return this.#pendingRewindReport; for (let i = messages.length - 1; i >= 0; i--) { const message = messages[i]; - if (message?.role !== "toolResult" || message.toolName !== "rewind" || message.isError) continue; - const details = message.details; + if (message?.role !== "toolResult" || message.isError) continue; + const semanticResult = semanticToolResult(message.toolName, message); + if (semanticResult?.toolName !== "rewind") continue; + const details = semanticResult.details; const detailReport = details && typeof details === "object" && "report" in details && typeof details.report === "string" ? details.report.trim() @@ -11548,16 +12561,18 @@ export class AgentSession { if (!checkpointState) { return; } - try { - this.sessionManager.branchWithSummary(checkpointState.checkpointEntryId, report, { - startedAt: checkpointState.startedAt, - }); - } catch (error) { - logger.warn("Rewind branch checkpoint missing, falling back to root", { - error: error instanceof Error ? error.message : String(error), - }); - this.sessionManager.branchWithSummary(null, report, { startedAt: checkpointState.startedAt }); - } + this.#withBashBranchTransition(() => { + try { + this.sessionManager.branchWithSummary(checkpointState.checkpointEntryId, report, { + startedAt: checkpointState.startedAt, + }); + } catch (error) { + logger.warn("Rewind branch checkpoint missing, falling back to root", { + error: error instanceof Error ? error.message : String(error), + }); + this.sessionManager.branchWithSummary(null, report, { startedAt: checkpointState.startedAt }); + } + }); const rewoundAt = new Date().toISOString(); const details = { report, startedAt: checkpointState.startedAt, rewoundAt }; @@ -11572,7 +12587,7 @@ export class AgentSession { if (activeMessages) { for (const message of activeMessages) { - if (message.role === "toolResult" && message.toolName === "rewind") { + if (message.role === "toolResult" && semanticToolResult(message.toolName, message)?.toolName === "rewind") { this.#rewoundToolResultIds.add(message.toolCallId); } } @@ -12871,12 +13886,15 @@ export class AgentSession { suppressContinuation?: boolean; suppressHandoff?: boolean; phase?: CodexCompactionContext["phase"]; + terminalTextAnswer?: boolean; } = {}, ): Promise { const compactionSettings = this.settings.getGroup("compaction"); if (compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; if (reason !== "idle" && !compactionSettings.enabled) return COMPACTION_CHECK_NONE; const generation = this.#promptGeneration; + const terminalTextAnswer = + options.terminalTextAnswer ?? isTerminalTextAssistantAnswer(this.#findLastAssistantMessage()); const suppressContinuation = options.suppressContinuation === true; const shouldAutoContinue = !suppressContinuation && options.autoContinue !== false && compactionSettings.autoContinue !== false; @@ -12891,6 +13909,7 @@ export class AgentSession { willRetry, generation, shouldAutoContinue, + terminalTextAnswer, options.triggerContextTokens, suppressContinuation, ); @@ -12913,11 +13932,17 @@ export class AgentSession { async signal => { await Promise.resolve(); if (signal.aborted) return; - await this.#runAutoCompaction(reason, willRetry, true, true, { phase: options.phase }); + await this.#runAutoCompaction(reason, willRetry, true, true, { + ...options, + terminalTextAnswer, + }); }, { generation }, ); - return COMPACTION_CHECK_DEFERRED_HANDOFF; + return { + ...COMPACTION_CHECK_DEFERRED_HANDOFF, + continuationScheduled: shouldAutoContinue, + }; } // "overflow" forces context-full because the input itself is broken — a handoff @@ -12984,10 +14009,14 @@ export class AgentSession { aborted: false, willRetry: false, }); - const continuationScheduled = !autoCompactionSignal.aborted && reason !== "idle" && shouldAutoContinue; - if (continuationScheduled) { - this.#scheduleAutoContinuePrompt(generation); - } + const continuationScheduled = + !autoCompactionSignal.aborted && + this.#scheduleCompactionContinuation({ + generation, + autoContinue: reason !== "idle" && shouldAutoContinue, + terminalTextAnswer, + suppressContinuation, + }); return { ...(continuationScheduled ? COMPACTION_CHECK_CONTINUATION : COMPACTION_CHECK_NONE), historyRewritten: true, @@ -13026,35 +14055,80 @@ export class AgentSession { this.#getCompactionModelCandidates(availableModels), this.sessionId, ); - const preparation = prepareCompaction(pathEntries, compactionSettings, autoCompactionCandidates); + let pathEntriesForCompaction = pathEntries; + let preparation = prepareCompaction(pathEntriesForCompaction, compactionSettings, autoCompactionCandidates); if (!preparation) { - await this.#emitSessionEvent({ - type: "auto_compaction_end", - action, - result: undefined, - aborted: false, - willRetry: false, - skipped: true, - }); - const noProgressDeadEnd = reason !== "idle"; - let continuationScheduled = false; - if (!suppressContinuation && this.agent.hasQueuedMessages()) { - this.#scheduleAgentContinue({ - delayMs: 100, - generation, - shouldContinue: () => this.agent.hasQueuedMessages(), + // prepareCompaction found nothing to summarize because the kept region + // is a single oversized recent turn — findCutPoint never cuts inside a + // tool result, so a huge tool-result / fenced block tail leaves nothing + // on the summarizable side and summary compaction cannot even start. + // That is exactly the dead-end the elide shake rescues: it reaches + // INSIDE the tail and offloads heavy content to an artifact placeholder, + // shrinking the tail so findCutPoint can then move the cut and leave + // older turns to summarize. Run the same tiered rescue the + // post-maintenance guard uses (elide, then image drop), with progress + // defined as "prepareCompaction now succeeds on the rewritten branch", + // and fall through to the normal compaction body when it does (writing + // a compaction entry anchors the stale billed usage so the + // auto-continue re-check cannot re-trip and loop the warning — issue + // #4786). `skipElide` when we already fell through from a shake + // strategy pass (it tried and found nothing); skip entirely on the + // idle timer (it re-checks usage on its own cadence). + let rescueRewroteHistory = false; + if (reason !== "idle") { + await this.#rescueCompactionDeadEnd(autoCompactionSignal, { + skipElide: fallbackFromShake, + hasProgress: () => { + // Only reached when a tier actually freed something, so the + // branch has been rewritten either way. + rescueRewroteHistory = true; + pathEntriesForCompaction = this.sessionManager.getBranch(); + preparation = prepareCompaction( + pathEntriesForCompaction, + compactionSettings, + autoCompactionCandidates, + ); + return preparation !== undefined; + }, }); - continuationScheduled = true; } - if (noProgressDeadEnd) { - this.emitNotice( - "warning", - compactionDeadEndWarning("shrink it (e.g. clear large tool output)"), - "compaction", - ); + if (!preparation) { + await this.#emitSessionEvent({ + type: "auto_compaction_end", + action, + result: undefined, + aborted: false, + willRetry: false, + skipped: true, + }); + const noProgressDeadEnd = reason !== "idle"; + let continuationScheduled = false; + if (!suppressContinuation && this.agent.hasQueuedMessages()) { + this.#scheduleAgentContinue({ + delayMs: 100, + generation, + shouldContinue: () => this.agent.hasQueuedMessages(), + }); + continuationScheduled = true; + } + if (noProgressDeadEnd) { + this.emitNotice( + "warning", + compactionDeadEndWarning("shrink it (e.g. clear large tool output)"), + "compaction", + ); + } + // A rescue that offloaded content but still could not produce a + // preparation rewrote the branch; flag it so the overflow-recovery + // rollback does not re-restore the just-failed assistant turn on top + // of the elided tail. + const base = continuationScheduled + ? COMPACTION_CHECK_CONTINUATION + : noProgressDeadEnd + ? COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION + : COMPACTION_CHECK_NONE; + return rescueRewroteHistory ? { ...base, historyRewritten: true } : base; } - if (continuationScheduled) return COMPACTION_CHECK_CONTINUATION; - return noProgressDeadEnd ? COMPACTION_CHECK_BLOCK_AUTOMATIC_CONTINUATION : COMPACTION_CHECK_NONE; } let hookCompaction: CompactionResult | undefined; @@ -13066,7 +14140,7 @@ export class AgentSession { const hookResult = (await this.#extensionRunner.emit({ type: "session_before_compact", preparation, - branchEntries: pathEntries, + branchEntries: pathEntriesForCompaction, customInstructions: undefined, signal: autoCompactionSignal, })) as SessionBeforeCompactResult | undefined; @@ -13476,20 +14550,13 @@ export class AgentSession { if (retryFits) { this.#scheduleAgentContinue({ delayMs: 100, generation }); continuationScheduled = true; - } else if (hasHeadroom && shouldAutoContinue) { - this.#scheduleAutoContinuePrompt(generation); - continuationScheduled = true; - } - if (!continuationScheduled && !suppressContinuation && this.agent.hasQueuedMessages()) { - // Auto-compaction can complete while follow-up/steering/custom messages are waiting. - // Kick the loop so queued messages are actually delivered. This remains separate - // from the no-progress warning: pausing maintenance must not strand user input. - this.#scheduleAgentContinue({ - delayMs: 100, + } else { + continuationScheduled = this.#scheduleCompactionContinuation({ generation, - shouldContinue: () => this.agent.hasQueuedMessages(), + autoContinue: hasHeadroom && shouldAutoContinue, + terminalTextAnswer, + suppressContinuation, }); - continuationScheduled = true; } if (deadEndWarning) { @@ -13545,6 +14612,7 @@ export class AgentSession { willRetry: boolean, generation: number, autoContinue: boolean, + terminalTextAnswer: boolean, triggerContextTokens?: number, suppressContinuation = false, ): Promise { @@ -13627,10 +14695,6 @@ export class AgentSession { }); let continuationScheduled = false; - if (!willRetry && reason !== "idle" && autoContinue) { - this.#scheduleAutoContinuePrompt(generation); - continuationScheduled = true; - } if (willRetry) { // The shake rebuild replays every entry, so a trailing error/length // assistant from the failed turn re-enters agent state — drop it before @@ -13646,13 +14710,13 @@ export class AgentSession { } this.#scheduleAgentContinue({ delayMs: 100, generation }); continuationScheduled = true; - } else if (!suppressContinuation && this.agent.hasQueuedMessages()) { - this.#scheduleAgentContinue({ - delayMs: 100, + } else { + continuationScheduled = this.#scheduleCompactionContinuation({ generation, - shouldContinue: () => this.agent.hasQueuedMessages(), + autoContinue: reason !== "idle" && autoContinue, + terminalTextAnswer, + suppressContinuation, }); - continuationScheduled = true; } if (!reclaimed) { return willRetry && continuationScheduled @@ -13790,6 +14854,50 @@ export class AgentSession { if (this.#isClassifierRefusal(message)) return true; return AIError.retriable(id, { replayUnsafe: this.#hasReplayUnsafeToolOutput(message) }); } + + /** + * Resume a stalled Cursor turn after every server-executed tool has produced + * a result. The failed assistant/tool-result pair must stay in context: it + * records completed side effects and lets the next request continue from + * them instead of replaying the original turn. + */ + #canResumeCursorStreamStall(message: AssistantMessage): boolean { + if ( + message.provider !== "cursor" || + message.stopReason !== "error" || + !message.errorMessage?.toLowerCase().includes("stream stall") + ) { + return false; + } + const id = this.#classifyRetryMessage(message); + if (!AIError.retriable(id)) return false; + + const resolvedToolCallIds: string[] = []; + for (const block of message.content) { + if (block.type !== "toolCall") continue; + if (!(kCursorExecResolved in block) || block[kCursorExecResolved] !== true) return false; + resolvedToolCallIds.push(block.id); + } + if (resolvedToolCallIds.length === 0) return false; + + const messages = this.agent.state.messages; + let assistantIndex = -1; + for (let i = messages.length - 1; i >= 0; i--) { + const candidate = messages[i]; + if (candidate.role === "assistant" && this.#isSameAssistantMessage(candidate, message)) { + assistantIndex = i; + break; + } + } + if (assistantIndex < 0) return false; + + const unresolvedToolCallIds = new Set(resolvedToolCallIds); + for (let i = assistantIndex + 1; i < messages.length; i++) { + const candidate = messages[i]; + if (candidate.role === "toolResult") unresolvedToolCallIds.delete(candidate.toolCallId); + } + return unresolvedToolCallIds.size === 0; + } /** * Retried turns remove the failed assistant message from active context. * Text/thinking-only partials are safe to discard and replay. Retained @@ -13800,12 +14908,36 @@ export class AgentSession { return message.content.some(block => block.type === "toolCall"); } + /** + * OpenRouter can repeatedly close Gemini streams at the reasoning-to-payload + * transition. One retry covers a transient edge failure; the normal ten-retry + * budget would otherwise re-run the same expensive reasoning cycle unchanged. + */ + #isOpenRouterThinkingStreamClose(message: AssistantMessage): boolean { + return ( + message.provider === "openrouter" && + /server_error:\s*stream closed with reason:\s*error/i.test(message.errorMessage ?? "") && + message.content.some(block => block.type === "thinking" && block.thinking.trim().length > 0) + ); + } + #isClassifierRefusal(message: AssistantMessage): boolean { if (message.stopReason !== "error") return false; const stopType = message.stopDetails?.type; return stopType === "refusal" || stopType === "sensitive"; } + /** + * True when `provider` has registered models or is configured for dynamic + * discovery. Discovery-only providers (e.g. a models.yml provider with + * `discovery:` and no static models) can hold zero models until the online + * refresh completes, so a models-only check would misreport them as + * unknown during session construction. + */ + #isKnownProvider(provider: string): boolean { + return this.#modelRegistry.hasProvider(provider); + } + #getRetryFallbackChains(): RetryFallbackChains { const configuredChains = this.settings.get("retry.fallbackChains"); if (!configuredChains || typeof configuredChains !== "object") return {}; @@ -13836,8 +14968,8 @@ export class AgentSession { const keyKind = isRetryFallbackModelKey(key) ? "model" : "role"; if (keyKind === "model") { if (isRetryFallbackWildcardKey(key)) { - const provider = key.slice(0, -2); - if (!this.#modelRegistry.getAll().some(model => model.provider === provider)) { + const { provider } = parseRetryFallbackWildcard(key, p => this.#isKnownProvider(p)); + if (!this.#isKnownProvider(provider)) { const msg = `retry.fallbackChains wildcard key references unknown provider: ${key}`; logger.warn(msg); this.configWarnings.push(msg); @@ -13869,8 +15001,8 @@ export class AgentSession { continue; } if (isRetryFallbackWildcardKey(selectorStr)) { - const provider = selectorStr.slice(0, -2); - if (!this.#modelRegistry.getAll().some(model => model.provider === provider)) { + const { provider } = parseRetryFallbackWildcard(selectorStr, p => this.#isKnownProvider(p)); + if (!this.#isKnownProvider(provider)) { const msg = `Fallback chain for ${keyKind} '${key}' references unknown provider: ${selectorStr}`; logger.warn(msg); this.configWarnings.push(msg); @@ -13929,13 +15061,16 @@ export class AgentSession { * Model-oriented keys win over roles so a chain follows the model across * role reassignments. */ - #resolveRetryFallbackRole(currentSelector: string): string | undefined { + #resolveRetryFallbackRole( + currentSelector: string, + currentModel: Model | null | undefined = this.model, + ): string | undefined { const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); if (!parsedCurrent) return undefined; const chains = this.#getRetryFallbackChains(); const currentBaseSelector = formatRetryFallbackBaseSelector(parsedCurrent); - const currentPlainSelector = this.model - ? formatModelSelectorValue(formatModelString(this.model), parsedCurrent.thinkingLevel) + const currentPlainSelector = currentModel + ? formatModelSelectorValue(formatModelString(currentModel), parsedCurrent.thinkingLevel) : undefined; const currentPlainBaseSelector = currentPlainSelector && currentPlainSelector !== currentSelector @@ -13961,9 +15096,22 @@ export class AgentSession { for (const key of exactModelKeys) { if (matchesCurrent(this.#getRetryFallbackPrimarySelector(key))) return key; } - // 2. Provider wildcard (`provider/*`) — any active model of this provider. - const wildcardKey = `${parsedCurrent.provider}/*`; - if (Array.isArray(chains[wildcardKey])) return wildcardKey; + // 2. Provider wildcards — an id-prefixed key (`openrouter/google/*`) + // beats the plain `provider/*` key for ids under its prefix. + let wildcardMatch: string | undefined; + let wildcardPrefixLength = -1; + for (const key in chains) { + if (!isRetryFallbackWildcardKey(key) || !Array.isArray(chains[key])) continue; + const { provider, idPrefix } = parseRetryFallbackWildcard(key, p => this.#isKnownProvider(p)); + if (provider !== parsedCurrent.provider) continue; + if (idPrefix !== undefined && !parsedCurrent.id.startsWith(`${idPrefix}/`)) continue; + const prefixLength = idPrefix === undefined ? 0 : idPrefix.length; + if (prefixLength > wildcardPrefixLength) { + wildcardMatch = key; + wildcardPrefixLength = prefixLength; + } + } + if (wildcardMatch) return wildcardMatch; // 3. Role keys — matched by the role's currently-assigned model. for (const key of roleKeys) { if (matchesCurrent(this.#getRetryFallbackPrimarySelector(key))) return key; @@ -13982,9 +15130,11 @@ export class AgentSession { /** * Parse one configured chain entry. A `provider/*` entry keeps the failing - * model's id and swaps the provider (google-antigravity/x → google/x); - * ids the target provider lacks are skipped by the candidate loop's - * registry lookup. + * model's id and swaps the provider (google-antigravity/x → google/x); an + * id-prefixed `provider/prefix/*` entry re-prefixes the failing model's + * bare id instead (openrouter/google/* : google-antigravity/x → + * openrouter/google/x). Ids the target provider lacks are skipped by the + * candidate loop's registry lookup. */ #parseRetryFallbackChainEntry( entry: string, @@ -13992,8 +15142,23 @@ export class AgentSession { ): RetryFallbackSelector | undefined { if (isRetryFallbackWildcardKey(entry)) { if (!current) return undefined; - const provider = entry.slice(0, -2); - return { raw: `${provider}/${current.id}`, provider, id: current.id, thinkingLevel: undefined }; + const { provider, idPrefix } = parseRetryFallbackWildcard(entry, p => this.#isKnownProvider(p)); + const bareId = current.id.slice(current.id.lastIndexOf("/") + 1); + let id: string; + if (idPrefix !== undefined) { + id = `${idPrefix}/${bareId}`; + } else if ( + bareId !== current.id && + !this.#modelRegistry.find(provider, current.id) && + this.#modelRegistry.find(provider, bareId) + ) { + // Aggregator → direct: the failing id carries a vendor prefix the + // target provider does not use (openrouter/google/x → google-vertex/x). + id = bareId; + } else { + id = current.id; + } + return { raw: `${provider}/${id}`, provider, id, thinkingLevel: undefined }; } return parseRetryFallbackSelector(entry, this.#modelRegistry); } @@ -14026,7 +15191,11 @@ export class AgentSession { return chain; } - #findRetryFallbackCandidates(role: string, currentSelector: string): RetryFallbackSelector[] { + #findRetryFallbackCandidates( + role: string, + currentSelector: string, + currentModel: Model | null | undefined = this.model, + ): RetryFallbackSelector[] { let chain = this.#getRetryFallbackEffectiveChain(role, currentSelector); const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); if (chain.length === 0 && role === "default" && parsedCurrent) { @@ -14050,8 +15219,8 @@ export class AgentSession { if (chain.length <= 1) return []; const currentBaseSelector = parsedCurrent ? formatRetryFallbackBaseSelector(parsedCurrent) : undefined; const currentPlainSelector = - this.model && parsedCurrent - ? formatModelSelectorValue(formatModelString(this.model), parsedCurrent.thinkingLevel) + currentModel && parsedCurrent + ? formatModelSelectorValue(formatModelString(currentModel), parsedCurrent.thinkingLevel) : undefined; const currentPlainBaseSelector = parsedCurrent && currentPlainSelector && currentPlainSelector !== currentSelector @@ -14325,7 +15494,12 @@ export class AgentSession { */ async #handleRetryableError( message: AssistantMessage, - options?: { allowModelFallback?: boolean; fireworksFastFallback?: boolean; hardErrorFallback?: boolean }, + options?: { + allowModelFallback?: boolean; + fireworksFastFallback?: boolean; + hardErrorFallback?: boolean; + preserveFailedTurn?: boolean; + }, ): Promise { const retrySettings = this.settings.getGroup("retry"); // The Fireworks Fast→base degrade is an intrinsic model-selection safety net, @@ -14351,7 +15525,10 @@ export class AgentSession { // (every rotation sets switchedCredential and skips it), so without // this last resort a provider-wide usage cap never fails over to the // configured chain. - const retryBudgetExhausted = this.#retryAttempt > retrySettings.maxRetries; + const maxRetries = this.#isOpenRouterThinkingStreamClose(message) + ? Math.min(retrySettings.maxRetries, 1) + : retrySettings.maxRetries; + const retryBudgetExhausted = this.#retryAttempt > maxRetries; const errorMessage = message.errorMessage || "Unknown error"; const id = this.#classifyRetryMessage(message); @@ -14444,7 +15621,7 @@ export class AgentSession { } if (retryBudgetExhausted) { if (!switchedModel) { - await this.#persistRetryLifecycleErrorMessage(message); + await this.#persistTerminalEmptyErrorTurn(message); // Max retries exceeded and no fallback model to switch to: emit // final failure and reset. await this.#emitSessionEvent({ @@ -14491,7 +15668,7 @@ export class AgentSession { // can act on it. const maxDelayMs = retrySettings.maxDelayMs; if (maxDelayMs > 0 && delayMs > maxDelayMs && !switchedCredential && !switchedModel) { - await this.#persistRetryLifecycleErrorMessage(message); + await this.#persistTerminalEmptyErrorTurn(message); const attempt = this.#retryAttempt; this.#retryAttempt = 0; await this.#emitSessionEvent({ @@ -14510,14 +15687,17 @@ export class AgentSession { await this.#emitSessionEvent({ type: "auto_retry_start", attempt: this.#retryAttempt, - maxAttempts: retrySettings.maxRetries, + maxAttempts: maxRetries, delayMs, errorMessage, errorId: message.errorId, }); - // Remove the failed assistant message from active context before retrying. - this.#removeAssistantMessageFromActiveContext(message, "auto-retry"); + // Cursor exec-channel tools have already run and emitted results. Keep that + // failed turn intact so continuation cannot repeat their side effects. + if (!options?.preserveFailedTurn) { + this.#removeAssistantMessageFromActiveContext(message, "auto-retry"); + } // A thinking/response loop retried into identical context loops again. Inject a // hidden redirect so the retried turn sees a directive to break the repeated @@ -14632,21 +15812,40 @@ export class AgentSession { } /** * Manually retry the last failed assistant turn. - * Removes the error message from agent state and re-attempts with a fresh retry budget. + * Removes the error message from active agent state when present and + * re-attempts with a fresh retry budget. + * + * A stream that stalls or aborts mid-tool-call ends the turn with + * `stopReason: "error" | "aborted"` and then appends one synthetic + * {@link isSyntheticToolResultMessage tool_result} per emitted tool call to + * preserve the provider's tool_use/tool_result pairing (see + * `createAbortedToolResult` in `agent-loop.ts`). Those placeholders trail the + * failed assistant turn, so the retry lookback walks back over them before + * checking the assistant message; it strips both the placeholders and the + * failed turn before re-attempting. + * + * A restored session deliberately omits failed assistant turns from provider + * context. In that case, the persisted display transcript remains the source + * of truth for whether the current branch has a retryable failed tail. + * * @returns true if retry was initiated, false if no failed turn to retry or agent is busy */ async retry(): Promise { if (this.isStreaming || this.isCompacting || this.isRetrying) return false; const messages = this.agent.state.messages; - const lastMsg = messages[messages.length - 1]; - if (lastMsg?.role !== "assistant") return false; - - const assistantMsg = lastMsg as AssistantMessage; - if (assistantMsg.stopReason !== "error" && assistantMsg.stopReason !== "aborted") return false; - - // Remove the failed/aborted assistant message (same as auto-retry does before re-attempting) - this.agent.replaceMessages(messages.slice(0, -1)); + const activeTurnEnd = retryableAssistantTurnEnd(messages); + if (activeTurnEnd !== undefined) { + // Remove the failed/aborted assistant message plus its synthetic tool + // results (same as auto-retry does before re-attempting). + this.agent.replaceMessages(messages.slice(0, activeTurnEnd - 1)); + } else { + // A restored session already dropped the failed assistant turn (and its + // paired synthetic tool results) from provider context, so the persisted + // display transcript is the source of truth for a retryable failed tail. + const transcriptMessages = this.sessionManager.buildSessionContext({ transcript: true }).messages; + if (retryableAssistantTurnEnd(transcriptMessages) === undefined) return false; + } // Reset retry budget for a fresh attempt this.#retryAttempt = 0; @@ -14661,71 +15860,22 @@ export class AgentSession { // Bash Execution // ========================================================================= - async #saveBashOriginalArtifact(originalText: string): Promise { + async #saveBashOriginalArtifact(target: BashSessionTarget, originalText: string): Promise { try { - return await this.sessionManager.saveArtifact(originalText, "bash-original"); + const destination = target.destination ?? (await target.pending); + return await destination?.manager.saveArtifact(originalText, "bash-original"); } catch { return undefined; } } - /** - * Execute a bash command. - * Adds result to agent context and session. - * @param command The bash command to execute - * @param onChunk Optional streaming callback for output - * @param options.excludeFromContext If true, command output won't be sent to LLM (!! prefix) - * @param options.useUserShell If true, allow caller to request configured user-shell routing - */ - async executeBash( + #createBashMessage( command: string, - onChunk?: (chunk: string) => void, - options?: { excludeFromContext?: boolean; useUserShell?: boolean }, - ): Promise { - const excludeFromContext = options?.excludeFromContext === true; - const cwd = this.sessionManager.getCwd(); - - if (this.#extensionRunner?.hasHandlers("user_bash")) { - const hookResult = await this.#extensionRunner.emitUserBash({ - type: "user_bash", - command, - excludeFromContext, - cwd, - }); - if (hookResult?.result) { - this.recordBashResult(command, hookResult.result, options); - return hookResult.result; - } - } - - const abortController = new AbortController(); - this.#bashAbortControllers.add(abortController); - - try { - const result = await executeBashCommand(command, { - onChunk, - signal: abortController.signal, - sessionKey: this.sessionId, - cwd, - timeout: clampTimeout("bash") * 1000, - onMinimizedSave: originalText => this.#saveBashOriginalArtifact(originalText), - useUserShell: options?.useUserShell, - }); - - this.recordBashResult(command, result, options); - return result; - } finally { - this.#bashAbortControllers.delete(abortController); - } - } - - /** - * Record a bash execution result in session history. - * Used by executeBash and by extensions that handle bash execution themselves. - */ - recordBashResult(command: string, result: BashResult, options?: { excludeFromContext?: boolean }): void { + result: BashResult, + options?: { excludeFromContext?: boolean }, + ): BashExecutionMessage { const meta = outputMeta().truncationFromSummary(result, { direction: "tail" }).get(); - const bashMessage: BashExecutionMessage = { + return { role: "bashExecution", command, output: result.output, @@ -14736,20 +15886,239 @@ export class AgentSession { timestamp: Date.now(), excludeFromContext: options?.excludeFromContext, }; + } - // If agent is streaming, defer adding to avoid breaking tool_use/tool_result ordering - if (this.isStreaming) { - // Queue for later - will be flushed on agent_end - this.#pendingBashMessages.push(bashMessage); - } else { - // Add to agent state immediately - this.agent.appendMessage(bashMessage); + #captureBashSessionTarget(): BashSessionTarget { + this.#bashSessionTarget.refs++; + return this.#bashSessionTarget; + } - // Save to session - this.sessionManager.appendMessage(bashMessage); + async #releaseBashSessionTarget(target: BashSessionTarget): Promise { + if (target.refs <= 0) throw new Error("Bash session target released more than once"); + target.refs--; + if (target.refs === 0 && target.destination?.kind === "detached") { + await target.destination.manager.close(); } } + #appendBashMessage(destination: BashAppendDestination, message: BashExecutionMessage): void { + switch (destination.kind) { + case "current": + this.agent.appendMessage(message); + destination.manager.appendMessage(message); + break; + case "detached": + destination.manager.appendMessage(message); + break; + case "branch": + destination.parentId = destination.manager.appendMessageToBranch(message, destination.parentId); + break; + } + } + + async #appendOwnedBashMessage(target: BashSessionTarget, message: BashExecutionMessage): Promise { + try { + const destination = target.destination ?? (await target.pending); + if (!destination) throw new Error("Bash session target has no append destination"); + this.#appendBashMessage(destination, message); + } finally { + await this.#releaseBashSessionTarget(target); + } + } + + async #recordBashResultForTarget( + target: BashSessionTarget, + command: string, + result: BashResult, + options?: { excludeFromContext?: boolean }, + ): Promise { + const message = this.#createBashMessage(command, result, options); + if (this.isStreaming && target === this.#bashSessionTarget) { + this.#pendingBashMessages.push({ target, message }); + return; + } + await this.#appendOwnedBashMessage(target, message); + } + + /** Run a leaf rewrite while retaining any in-flight bash on its originating branch. */ + #withBashBranchTransition(mutate: () => T): T { + const bashTransition = this.#beginBashSessionTransition(); + let branchTransitioned = false; + try { + const result = mutate(); + this.#markBashSessionTransition(bashTransition); + branchTransitioned = true; + return result; + } finally { + this.#finishBashSessionTransition(bashTransition, branchTransitioned); + } + } + + /** + * Snapshot the session/branch that owns any in-flight bash before a transition. + * When an owner is still active, its target is detached to a clone so a failed + * or intentionally dropped transition never redirects the late result. + */ + #beginBashSessionTransition(options?: { persistDetached?: boolean }): BashSessionTransition { + const oldTarget = this.#bashSessionTarget; + let detachedManager: SessionManager | undefined; + let resolveOld: ((destination: BashAppendDestination) => void) | undefined; + if (oldTarget.refs > 0) { + detachedManager = this.sessionManager.cloneCurrentSession({ persist: options?.persistDetached }); + const pendingOld = Promise.withResolvers(); + oldTarget.destination = undefined; + oldTarget.pending = pendingOld.promise; + resolveOld = pendingOld.resolve; + } + + const pendingNew = Promise.withResolvers(); + return { + oldTarget, + newTarget: { + sessionId: this.sessionManager.getSessionId(), + refs: 0, + pending: pendingNew.promise, + }, + oldSessionId: this.sessionManager.getSessionId(), + oldSessionFile: this.sessionManager.getSessionFile(), + oldLeafId: this.sessionManager.getLeafId(), + detachedManager, + resolveOld, + resolveNew: pendingNew.resolve, + }; + } + + /** Adopt the transition's new target as the live bash owner. */ + #markBashSessionTransition(transition: BashSessionTransition): void { + transition.newTarget.sessionId = this.sessionManager.getSessionId(); + this.#bashSessionTarget = transition.newTarget; + } + + /** + * Resolve the pending append destinations opened by {@link #beginBashSessionTransition}. + * On success the old owner keeps its original session/branch (same file → current or + * branch destination; different file → detached clone); on failure both fall back to + * the still-current manager and the clone is discarded. + */ + #finishBashSessionTransition(transition: BashSessionTransition, success: boolean): void { + const currentDestination: BashAppendDestination = { kind: "current", manager: this.sessionManager }; + let oldDestination: BashAppendDestination = currentDestination; + if (success && transition.resolveOld) { + const currentFile = this.sessionManager.getSessionFile(); + const sameFile = + transition.oldSessionFile === currentFile || + (transition.oldSessionFile !== undefined && + currentFile !== undefined && + path.resolve(transition.oldSessionFile) === path.resolve(currentFile)); + const sameSession = transition.oldSessionId === this.sessionManager.getSessionId() && sameFile; + if (sameSession) { + oldDestination = + transition.oldLeafId === this.sessionManager.getLeafId() + ? currentDestination + : { kind: "branch", manager: this.sessionManager, parentId: transition.oldLeafId }; + } else if (transition.detachedManager) { + oldDestination = { kind: "detached", manager: transition.detachedManager }; + } + } + + if (transition.resolveOld) { + transition.oldTarget.pending = undefined; + transition.oldTarget.destination = oldDestination; + transition.resolveOld(oldDestination); + } + + transition.newTarget.pending = undefined; + transition.newTarget.destination = currentDestination; + if (!success) transition.newTarget.sessionId = this.sessionManager.getSessionId(); + transition.resolveNew(currentDestination); + + if (transition.detachedManager && (oldDestination.kind !== "detached" || transition.oldTarget.refs === 0)) { + void transition.detachedManager.close().catch(error => { + logger.warn("Failed to close detached bash session writer", { error: String(error) }); + }); + } + } + + /** + * Execute a bash command and retain the session/branch that owned its start. + * @param command The bash command to execute + * @param onChunk Optional streaming callback for output + * @param options.excludeFromContext If true, command output won't be sent to LLM (!! prefix) + * @param options.useUserShell If true, allow caller to request configured user-shell routing + */ + async executeBash( + command: string, + onChunk?: (chunk: string) => void, + options?: { excludeFromContext?: boolean; useUserShell?: boolean }, + ): Promise { + const target = this.#captureBashSessionTarget(); + let targetTransferred = false; + const excludeFromContext = options?.excludeFromContext === true; + const cwd = this.sessionManager.getCwd(); + + try { + if (this.#extensionRunner?.hasHandlers("user_bash")) { + const hookResult = await this.#extensionRunner.emitUserBash({ + type: "user_bash", + command, + excludeFromContext, + cwd, + }); + if (hookResult?.result) { + targetTransferred = true; + await this.#recordBashResultForTarget(target, command, hookResult.result, options); + return hookResult.result; + } + } + + const abortController = new AbortController(); + this.#bashAbortControllers.add(abortController); + let result: BashResult; + try { + result = await executeBashCommand(command, { + onChunk, + signal: abortController.signal, + sessionKey: target.sessionId, + cwd, + timeout: clampTimeout("bash", undefined, this.settings.get("tools.maxTimeout")) * 1000, + onMinimizedSave: originalText => this.#saveBashOriginalArtifact(target, originalText), + useUserShell: options?.useUserShell, + }); + } finally { + this.#bashAbortControllers.delete(abortController); + } + + targetTransferred = true; + await this.#recordBashResultForTarget(target, command, result, options); + return result; + } finally { + if (!targetTransferred) await this.#releaseBashSessionTarget(target); + } + } + + /** Record a bash result supplied outside executeBash in the current ownership scope. */ + recordBashResult(command: string, result: BashResult, options?: { excludeFromContext?: boolean }): void { + const target = this.#captureBashSessionTarget(); + const message = this.#createBashMessage(command, result, options); + if (this.isStreaming && target === this.#bashSessionTarget) { + this.#pendingBashMessages.push({ target, message }); + return; + } + + if (target.destination) { + try { + this.#appendBashMessage(target.destination, message); + } finally { + void this.#releaseBashSessionTarget(target); + } + return; + } + + void this.#appendOwnedBashMessage(target, message).catch(error => { + logger.error("Failed to record bash result in its owning session", { error: String(error) }); + }); + } + /** * Cancel running bash command. */ @@ -14769,22 +16138,14 @@ export class AgentSession { return this.#pendingBashMessages.length > 0; } - /** - * Flush pending bash messages to agent state and session. - * Called after agent turn completes to maintain proper message ordering. - */ - #flushPendingBashMessages(): void { + /** Flush pending bash messages after the active turn without changing their ownership. */ + async #flushPendingBashMessages(): Promise { if (this.#pendingBashMessages.length === 0) return; - - for (const bashMessage of this.#pendingBashMessages) { - // Add to agent state - this.agent.appendMessage(bashMessage); - - // Save to session - this.sessionManager.appendMessage(bashMessage); - } - + const pending = this.#pendingBashMessages; this.#pendingBashMessages = []; + for (const { target, message } of pending) { + await this.#appendOwnedBashMessage(target, message); + } } // ========================================================================= @@ -15377,9 +16738,11 @@ export class AgentSession { this.#disconnectFromAgent(); await this.abort({ goalReason: "internal" }); + await this.#flushPendingBashMessages(); // Flush pending writes before switching so restore snapshots reflect committed state. await this.sessionManager.flush(); const previousSessionState = this.sessionManager.captureState(); + const bashTransition = this.#beginBashSessionTransition(); // Only same-session reloads compare against the prior context to detect // rollback edits (`#didSessionMessagesChange` below). Building it for a // different-session switch is a pure waste — and on huge pre-fix sessions @@ -15424,6 +16787,7 @@ export class AgentSession { try { await this.sessionManager.setSessionFile(sessionPath); + this.#markBashSessionTransition(bashTransition); if (switchingToDifferentSession) { this.#freshProviderSessionId = undefined; this.#clearInheritedProviderPromptCacheKey(); @@ -15557,6 +16921,7 @@ export class AgentSession { error: String(error), }); } + this.#finishBashSessionTransition(bashTransition, true); return true; } catch (error) { this.sessionManager.restoreState(previousSessionState); @@ -15588,6 +16953,7 @@ export class AgentSession { this.#syncTodoPhasesFromBranch(); this.#resetAllAdvisorRuntimes(); this.#reconnectToAgent(); + this.#finishBashSessionTransition(bashTransition, false); throw error; } } @@ -15633,14 +16999,25 @@ export class AgentSession { this.#pendingNextTurnMessages = []; this.#scheduledHiddenNextTurnGeneration = undefined; + await this.#flushPendingBashMessages(); // Flush pending writes before branching await this.sessionManager.flush(); + const bashTransition = this.#beginBashSessionTransition(); this.#cancelOwnAsyncJobs(); + this.#abortAutolearnCapture(); + await this.#drainAutolearnCapture(); - if (!selectedEntry.parentId) { - await this.sessionManager.newSession({ parentSession: previousSessionFile }); - } else { - this.sessionManager.createBranchedSession(selectedEntry.parentId); + let sessionTransitioned = false; + try { + if (!selectedEntry.parentId) { + await this.sessionManager.newSession({ parentSession: previousSessionFile }); + } else { + this.sessionManager.createBranchedSession(selectedEntry.parentId); + } + this.#markBashSessionTransition(bashTransition); + sessionTransitioned = true; + } finally { + this.#finishBashSessionTransition(bashTransition, sessionTransitioned); } this.#rehydrateCheckpointRewindState(); this.#syncTodoPhasesFromBranch(); @@ -15724,10 +17101,21 @@ export class AgentSession { await this.abort({ goalReason: "internal", reason: "branching /btw" }); this.agent.replaceQueues([], []); } + await this.#flushPendingBashMessages(); await this.sessionManager.flush(); + const bashTransition = this.#beginBashSessionTransition(); this.#cancelOwnAsyncJobs(); + this.#abortAutolearnCapture(); + await this.#drainAutolearnCapture(); - this.sessionManager.createBranchedSession(leafId); + let sessionTransitioned = false; + try { + this.sessionManager.createBranchedSession(leafId); + this.#markBashSessionTransition(bashTransition); + sessionTransitioned = true; + } finally { + this.#finishBashSessionTransition(bashTransition, sessionTransitioned); + } this.#rehydrateCheckpointRewindState(); this.sessionManager.appendMessage({ @@ -15774,7 +17162,30 @@ export class AgentSession { */ async navigateTree( targetId: string, - options: { summarize?: boolean; customInstructions?: string } = {}, + options: { + summarize?: boolean; + customInstructions?: string; + /** + * Opts into the two-phase `ask` toolResult re-answer protocol + * (issue #5642): set only by the interactive `/tree` selector, which + * knows how to re-open the picker on `reopenAsk` and complete the + * navigation with `reanswerAskResult`. Every other public caller + * (extensions, hooks, ACP, session-extension actions) leaves this + * unset and gets the pre-#5642 plain leaf move onto `ask` + * toolResults instead — they have no picker to re-open and would + * otherwise report a successful no-op navigation (roboomp review on + * #5895). + */ + allowAskReopen?: boolean; + /** + * Completes an in-progress `ask` re-answer (issue #5642): the caller + * already received `reopenAsk` from a prior call on the same + * `targetId`, re-opened the picker, and is handing back the fresh + * answer. Branches a new toolResult sibling instead of landing on + * the original one. + */ + reanswerAskResult?: AgentToolResult; + } = {}, ): Promise<{ editorText?: string; cancelled: boolean; @@ -15782,11 +17193,36 @@ export class AgentSession { summaryEntry?: BranchSummaryEntry; /** Raw session context built during navigation — pass to renderInitialMessages to skip a second O(N) walk. */ sessionContext?: SessionContext; + /** + * Set when `targetId` is an `ask` toolResult, `options.allowAskReopen` + * was set, and `options.reanswerAskResult` was not supplied: nothing was + * mutated. The caller must re-open the ask picker with these + * `questions`, then call `navigateTree(targetId, { ...options, + * reanswerAskResult })` with the produced result to actually branch + * (issue #5642). + */ + reopenAsk?: { toolCallId: string; questions: AskToolInput["questions"] }; }> { + await this.#flushPendingBashMessages(); const oldLeafId = this.sessionManager.getLeafId(); - // No-op if already at target - if (targetId === oldLeafId) { + const targetEntry = this.sessionManager.getEntry(targetId); + if (!targetEntry) { + throw new Error(`Entry ${targetId} not found`); + } + const targetIsAskResult = + targetEntry.type === "message" && + targetEntry.message.role === "toolResult" && + targetEntry.message.toolName === "ask"; + + // No-op if already at target — except mid-flight through the `ask` + // re-answer protocol (issue #5642): a probe or completion call can + // legitimately target the *current* leaf (e.g. the user interrupted + // right after answering `ask`, before a follow-up assistant message + // landed, or another caller navigated straight onto the ask result), + // and must still return `reopenAsk` / branch the new answer instead of + // silently reporting a no-op (chatgpt-codex review on #5895). + if (targetId === oldLeafId && !(options.allowAskReopen && targetIsAskResult)) { return { cancelled: false }; } @@ -15795,16 +17231,47 @@ export class AgentSession { throw new Error("No model available for summarization"); } - const targetEntry = this.sessionManager.getEntry(targetId); - if (!targetEntry) { - throw new Error(`Entry ${targetId} not found`); + // `ask` toolResult, first pass: hand control back to the caller to + // re-open the picker instead of landing on the stale answer in place. + // Nothing is mutated here — see the `reanswerAskResult` branch below for + // the actual sibling-branch construction once the caller has an answer. + // Gated on `allowAskReopen` — callers that don't understand `reopenAsk` + // fall straight through to the plain leaf move below instead of + // reporting a successful no-op (roboomp review on #5895). + if ( + options.allowAskReopen && + !options.reanswerAskResult && + targetEntry.type === "message" && + targetEntry.message.role === "toolResult" && + targetEntry.message.toolName === "ask" + ) { + const toolCallId = targetEntry.message.toolCallId; + const questions = this.#recoverAskReanswerQuestions(targetEntry.parentId, toolCallId); + if (questions) { + return { cancelled: false, reopenAsk: { toolCallId, questions } }; + } + // Original arguments couldn't be recovered (corrupted/legacy session + // data) — fall through to a plain leaf move so navigation still works. } - // Collect entries to summarize (from old leaf to common ancestor) + // Collect entries to summarize (from old leaf to common ancestor). For an + // `ask` re-answer completion, the branch point is `targetEntry.parentId` + // (the new sibling toolResult lands there, not on `targetId`) — anchor + // the collection there too, or the old answer entry is neither on the + // new branch nor included in the summary (chatgpt-codex review on + // #5895). + const summaryAnchorId = + options.reanswerAskResult !== undefined && + targetEntry.type === "message" && + targetEntry.message.role === "toolResult" && + targetEntry.message.toolName === "ask" && + targetEntry.parentId !== null + ? targetEntry.parentId + : targetId; const { entries: entriesToSummarize, commonAncestorId } = collectEntriesForBranchSummary( this.sessionManager, oldLeafId, - targetId, + summaryAnchorId, ); // Prepare event data @@ -15900,6 +17367,27 @@ export class AgentSession { .filter((c): c is { type: "text"; text: string } => c.type === "text") .map(c => c.text) .join(""); + } else if ( + targetEntry.type === "message" && + targetEntry.message.role === "toolResult" && + targetEntry.message.toolName === "ask" && + options.reanswerAskResult + ) { + // `ask` toolResult, second pass: the caller re-opened the picker and + // is handing back a fresh answer. Branch a *new* sibling toolResult + // off the same `ask` toolCall instead of reusing `targetId` — the + // original answer's branch stays reachable (issue #5642). + const reanswer = options.reanswerAskResult; + const toolResultMessage: ToolResultMessage = { + role: "toolResult", + toolCallId: targetEntry.message.toolCallId, + toolName: "ask", + content: reanswer.content, + details: reanswer.details, + isError: reanswer.isError === true, + timestamp: Date.now(), + }; + newLeafId = this.sessionManager.appendMessageToBranch(toolResultMessage, targetEntry.parentId); } else { // Non-user message (or a user-invoked skill-prompt injection): land the // leaf on the selected node so it stays on the active branch. Skill @@ -15910,18 +17398,28 @@ export class AgentSession { // Switch leaf (with or without summary) // Summary is attached at the navigation target position (newLeafId), not the old branch + const bashTransition = this.#beginBashSessionTransition(); let summaryEntry: BranchSummaryEntry | undefined; - if (summaryText) { - // Create summary at target position (can be null for root) - const summaryId = this.sessionManager.branchWithSummary(newLeafId, summaryText, summaryDetails, fromExtension); - - summaryEntry = this.sessionManager.getEntry(summaryId) as BranchSummaryEntry; - } else if (newLeafId === null) { - // No summary, navigating to root - reset leaf - this.sessionManager.resetLeaf(); - } else { - // No summary, navigating to non-root - this.sessionManager.branch(newLeafId); + let branchTransitioned = false; + try { + if (summaryText) { + // Create summary at target position (can be null for root) + const summaryId = this.sessionManager.branchWithSummary( + newLeafId, + summaryText, + summaryDetails, + fromExtension, + ); + summaryEntry = this.sessionManager.getEntry(summaryId) as BranchSummaryEntry; + } else if (newLeafId === null) { + this.sessionManager.resetLeaf(); + } else { + this.sessionManager.branch(newLeafId); + } + this.#markBashSessionTransition(bashTransition); + branchTransitioned = true; + } finally { + this.#finishBashSessionTransition(bashTransition, branchTransitioned); } // Update agent state — build display context to populate agent messages. @@ -15952,6 +17450,73 @@ export class AgentSession { return { editorText, cancelled: false, summaryEntry, sessionContext: stateContext }; } + /** + * Look up the `ask` toolCall's persisted `arguments` and validate them + * back into `questions`, for `/tree` `ask` re-answer (issue #5642). Walks + * up from the toolResult's parent past any interleaved ancestor entries + * — sibling toolResults from other tool calls in the same turn (`ask` + * runs `exclusive`, which only serializes *execution*, not persistence + * order — roboomp review on #5895), and bookkeeping entries such as the + * `tool_execution_start` custom entry `#recordToolExecutionStart()` + * appends before every toolResult in real persisted sessions (chatgpt-codex + * review on #5895) — until it finds the assistant entry that actually + * emitted `toolCallId`. Stops at a `user` message (turn boundary) or a + * dead end. Returns `undefined` when no ancestor entry holds a matching + * `ask` toolCall, or the arguments can't be resolved — the caller falls + * back to a plain leaf move rather than opening a picker with bad data. + */ + #recoverAskReanswerQuestions(parentId: string | null, toolCallId: string): AskToolInput["questions"] | undefined { + let current = parentId; + while (current !== null) { + const entry = this.sessionManager.getEntry(current); + if (!entry) return undefined; + if (entry.type === "message") { + if (entry.message.role === "assistant") { + const toolCall = entry.message.content.find( + (block): block is AgentToolCall => block.type === "toolCall" && block.id === toolCallId, + ); + if (!toolCall) return undefined; + if (toolCall.name !== "ask") return undefined; + const args = this.#obfuscator?.hasSecrets() + ? deobfuscateToolArguments(this.#obfuscator, toolCall.arguments) + : toolCall.arguments; + return recoverAskQuestions(args); + } + if (entry.message.role === "user") return undefined; + } + current = entry.parentId; + } + return undefined; + } + + /** + * Build a standalone `AgentToolContext` for running `AskTool.execute()` + * outside a normal agent turn, for `/tree` `ask` re-answer (issue #5642). + * `SelectorController` has no reachable `ToolContextStore` (that store is + * built inside `sdk.ts` and never threaded through to mode controllers), + * so this mirrors `refreshMCPTools()`'s `getCustomToolContext` factory + * with real session state instead of a `{ ... } as unknown as + * AgentToolContext` cast that could silently compile with an incomplete + * context (roboomp review on #5895) — every `CustomToolContext` field is + * backed by live session state, so a future required field fails to + * compile here instead of surfacing as `undefined` at runtime. + */ + buildAskReanswerContext(uiContext: ExtensionUIContext): AgentToolContext { + return { + sessionManager: this.sessionManager, + modelRegistry: this.#modelRegistry, + model: this.model, + isIdle: () => !this.isStreaming, + hasQueuedMessages: () => this.queuedMessageCount > 0, + abort: () => { + this.agent.abort(); + }, + settings: this.settings, + ui: uiContext, + hasUI: true, + }; + } + /** * Get all user messages from session for branch selector. */ @@ -16664,29 +18229,73 @@ export class AgentSession { return this.#advisors[0]?.agent; } + /** + * Lightweight advisor status for the status line: returns just the configured + * flag and per-advisor name/status without computing token/cost breakdowns. + * Avoids re-tokenizing the advisor transcript on every render frame. + */ + getAdvisorStatusOverview(): { configured: boolean; advisors: { name: string; status: AdvisorRuntimeStatus }[] } { + // Override stale map entries with live runtime status: failureNotified/quotaExhausted + // clear on reset() but #advisorStatuses lags until the next build. + const liveStatusBySlug = new Map(); + for (const a of this.#advisors) { + liveStatusBySlug.set( + a.slug, + a.runtime.quotaExhausted ? "quota_exhausted" : a.runtime.failureNotified ? "error" : "running", + ); + } + const advisors = [...this.#advisorStatuses.entries()].map(([slug, { name, status }]) => ({ + name, + status: liveStatusBySlug.get(slug) ?? status, + })); + return { configured: this.#advisorEnabled, advisors }; + } /** * Return structured advisor stats for the status command and TUI panel. */ getAdvisorStats(): AdvisorStats { const configured = this.#advisorEnabled; - const advisors = this.#advisors.map(a => this.#computeAdvisorStat(a)); - if (advisors.length === 0) { + const liveAdvisors = this.#advisors.map(a => this.#computeAdvisorStat(a)); + // Build the complete roster from #advisorStatuses, which already has the + // correct de-duped slugs as keys. Live advisors (from #advisors) carry full + // token/cost data; disabled/no-model/quota-exhausted advisors appear as + // skeleton entries with just name + status so the status line renders a dot. + const liveStatBySlug = new Map(this.#advisors.map((a, i) => [a.slug, liveAdvisors[i]])); + const roster: PerAdvisorStat[] = []; + for (const [slug, entry] of this.#advisorStatuses) { + const live = liveStatBySlug.get(slug); + if (live) { + roster.push(live); + } else { + roster.push({ + name: entry.name, + status: entry.status, + contextWindow: 0, + contextTokens: 0, + tokens: { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + cost: 0, + messages: { user: 0, assistant: 0, total: 0 }, + }); + } + } + const active = liveAdvisors.length > 0; + if (liveAdvisors.length === 0) { return { configured, - active: false, + active, contextWindow: 0, contextTokens: 0, tokens: { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, cost: 0, messages: { user: 0, assistant: 0, total: 0 }, - advisors: [], + advisors: roster, }; } const tokens = { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0, total: 0 }; const messages = { user: 0, assistant: 0, total: 0 }; let cost = 0; let contextTokens = 0; - for (const a of advisors) { + for (const a of liveAdvisors) { tokens.input += a.tokens.input; tokens.output += a.tokens.output; tokens.reasoning += a.tokens.reasoning; @@ -16703,14 +18312,14 @@ export class AgentSession { // first advisor's so the legacy status line stays byte-identical. return { configured, - active: true, - model: advisors[0].model, - contextWindow: advisors[0].contextWindow, + active, + model: liveAdvisors[0].model, + contextWindow: liveAdvisors[0].contextWindow, contextTokens, tokens, cost, messages, - advisors, + advisors: roster, }; } @@ -16744,12 +18353,18 @@ export class AgentSession { } return { name: advisor.name, + status: advisor.runtime.quotaExhausted + ? "quota_exhausted" + : advisor.runtime.failureNotified + ? "error" + : "running", model, contextWindow: model.contextWindow ?? 0, contextTokens, tokens: { input, output, reasoning, cacheRead, cacheWrite, total: totalTokens }, cost, messages: { user, assistant, total: messages.length }, + sessionId: advisor.agent.sessionId, }; } @@ -16758,13 +18373,18 @@ export class AgentSession { */ formatAdvisorStatus(): string { const stats = this.getAdvisorStats(); - if (!stats.active) { + if (!stats.active && stats.advisors.length === 0) { return stats.configured ? "Advisor setting is enabled, but no model is assigned to the 'advisor' role." : "Advisor is disabled."; } if (stats.advisors.length <= 1) { const s = stats.advisors[0]; + if (s && s.status === "no_model") { + return stats.configured + ? "Advisor setting is enabled, but no model is assigned to the 'advisor' role." + : "Advisor is disabled."; + } const contextLine = s.contextWindow > 0 ? `Context: ${s.contextTokens.toLocaleString()} / ${s.contextWindow.toLocaleString()} tokens (${Math.round((s.contextTokens / s.contextWindow) * 100)}%)` @@ -16773,6 +18393,7 @@ export class AgentSession { if (s.tokens.cacheRead > 0) spendParts.push(`${s.tokens.cacheRead.toLocaleString()} cache read`); if (s.tokens.cacheWrite > 0) spendParts.push(`${s.tokens.cacheWrite.toLocaleString()} cache write`); const spendLine = `Spend: ${spendParts.join(", ")}, $${s.cost.toFixed(4)}`; + if (!s.model || s.status !== "running") return `Advisor "${s.name}" is ${s.status.replace("_", " ")}.`; return `Advisor is enabled (${s.model.provider}/${s.model.id}). ${contextLine}. ${spendLine}.`; } const lines = [`Advisors enabled (${stats.advisors.length}):`]; @@ -16781,7 +18402,9 @@ export class AgentSession { s.contextWindow > 0 ? `${s.contextTokens.toLocaleString()} / ${s.contextWindow.toLocaleString()} (${Math.round((s.contextTokens / s.contextWindow) * 100)}%)` : `${s.contextTokens.toLocaleString()}`; - lines.push(` • ${s.name} (${s.model.provider}/${s.model.id}) — context ${ctx} tokens, $${s.cost.toFixed(4)}`); + lines.push( + ` • ${s.name}${s.model && s.status === "running" ? ` (${s.model.provider}/${s.model.id})` : ` [${s.status}]`} — context ${ctx} tokens, $${s.cost.toFixed(4)}`, + ); } lines.push( `Totals: ${stats.tokens.input.toLocaleString()} input, ${stats.tokens.output.toLocaleString()} output, $${stats.cost.toFixed(4)}.`, @@ -16790,37 +18413,51 @@ export class AgentSession { } /** - * Estimate the advisor's current context tokens. When the advisor has a - * recent non-aborted assistant message with usage, use that prompt's token - * count and add a trailing estimate for messages after it. Otherwise estimate - * every message. + * Estimate the advisor's current context tokens. A successful provider usage + * after the latest advisor compaction is ground truth for the prompt plus its + * generated output; only messages after that anchor are estimated. Usage from + * retained pre-compaction messages is stale and must not immediately retrigger + * maintenance on the newly compacted context. */ #estimateAdvisorContextTokens(messages: AgentMessage[]): number { - let lastUsageIndex: number | null = null; - let lastUsage: AssistantMessage["usage"] | undefined; + let usageAnchorStartIndex = 0; for (let i = messages.length - 1; i >= 0; i--) { - const msg = messages[i]; - if (msg.role === "assistant") { - const assistantMsg = msg as AssistantMessage; - if (assistantMsg.stopReason !== "aborted" && assistantMsg.stopReason !== "error" && assistantMsg.usage) { - lastUsage = assistantMsg.usage; - lastUsageIndex = i; - break; - } + const message = messages[i]; + if (message.role !== "compactionSummary") continue; + const advisorSummary = message as AdvisorCompactionSummaryMessage; + // Advisor summaries created before this runtime-only boundary existed have + // no trustworthy way to distinguish retained from newly appended messages. + // Conservatively ignore every current assistant until the next compaction. + usageAnchorStartIndex = advisorSummary.advisorUsageAnchorStartIndex ?? messages.length; + break; + } + + let lastUsageIndex: number | undefined; + let lastUsage: AssistantMessage["usage"] | undefined; + for (let i = messages.length - 1; i >= usageAnchorStartIndex; i--) { + const message = messages[i]; + if (message.role !== "assistant") continue; + const assistant = message as AssistantMessage; + if (assistant.stopReason !== "aborted" && assistant.stopReason !== "error" && assistant.usage) { + lastUsage = assistant.usage; + lastUsageIndex = i; + break; } } - if (!lastUsage || lastUsageIndex === null) { + + const estimateOptions = { excludeEncryptedReasoning: true } as const; + if (!lastUsage || lastUsageIndex === undefined) { let estimated = 0; for (const message of messages) { - estimated += estimateTokens(message); + estimated += estimateTokens(message, estimateOptions); } return estimated; } let trailingTokens = 0; for (let i = lastUsageIndex + 1; i < messages.length; i++) { - trailingTokens += estimateTokens(messages[i]); + trailingTokens += estimateTokens(messages[i], estimateOptions); } - return calculatePromptTokens(lastUsage) + trailingTokens; + return calculateContextTokens(lastUsage) + trailingTokens; } /** diff --git a/packages/coding-agent/src/session/messages.test.ts b/packages/coding-agent/src/session/messages.test.ts index cfc01d73b..5d6680047 100644 --- a/packages/coding-agent/src/session/messages.test.ts +++ b/packages/coding-agent/src/session/messages.test.ts @@ -8,6 +8,7 @@ import { replaceLlmImagesWithText, SKILL_PROMPT_MESSAGE_TYPE, type SkillPromptDetails, + stripImagesFromMessage, } from "./messages"; function customMessage(customType: string, attribution: "agent" | "user"): CustomMessage { @@ -125,6 +126,96 @@ describe("convertToLlm", () => { }); }); +function settledAssistant(text: string): AssistantMessage { + return { + role: "assistant", + content: [{ type: "text", text }], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-sonnet-4-5", + usage: { + input: 100, + output: 20, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 120, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 1, + }; +} + +function userMessage(text: string, timestamp: number): AgentMessage { + return { role: "user", content: text, attribution: "user", timestamp } as AgentMessage; +} + +describe("convertToLlm caching", () => { + it("reuses the outer array on an exact repeat of the same history", () => { + const messages: AgentMessage[] = [userMessage("hello", 1), settledAssistant("hi")]; + const first = convertToLlm(messages); + const second = convertToLlm(messages); + expect(second).toBe(first); + }); + + it("reuses the unchanged prefix output on append-only growth", () => { + const messages: AgentMessage[] = [userMessage("one", 1), settledAssistant("reply one")]; + const first = convertToLlm(messages); + messages.push(userMessage("two", 2)); + const grown = convertToLlm(messages); + // New outer array (no held-result aliasing), but the converted prefix is + // byte-identical and the appended turn is present. + expect(grown).not.toBe(first); + expect(grown.length).toBe(first.length + 1); + expect(grown.slice(0, first.length)).toEqual(first); + expect(grown[grown.length - 1]?.role).toBe("user"); + }); + + it("recomputes the boundary assistant when a following interrupted-thinking marker appears on growth", () => { + const messages: AgentMessage[] = [ + abortedAssistant([ + { type: "text", text: "partial answer" }, + { type: "thinking", thinking: "interrupted reasoning" }, + ]), + ]; + const before = convertToLlm(messages); + const beforeAssistant = before.find(entry => entry.role === "assistant"); + expect(Array.isArray(beforeAssistant?.content) && beforeAssistant.content.map(b => b.type)).toEqual([ + "text", + "thinking", + ]); + + // Append the continuity marker on the same array: the assistant is now the + // boundary message and its LLM view must drop the trailing thinking run. + messages.push(interruptedThinkingContinuity()); + const after = convertToLlm(messages); + const afterAssistant = after.find(entry => entry.role === "assistant"); + expect(Array.isArray(afterAssistant?.content) && afterAssistant.content.map(b => b.type)).toEqual(["text"]); + }); + + it("recomputes a message after strip-images invalidates its cache", () => { + const withImage: AgentMessage = { + role: "user", + content: [ + { type: "text", text: "look" }, + { type: "image", data: "aaaa", mimeType: "image/png" }, + ], + attribution: "user", + timestamp: 1, + }; + const messages: AgentMessage[] = [withImage]; + const before = convertToLlm(messages); + const beforeUser = before.find(entry => entry.role === "user"); + expect(Array.isArray(beforeUser?.content) && beforeUser.content.some(b => b.type === "image")).toBe(true); + + // Mutate in place through the owner seam, which must invalidate the cache. + stripImagesFromMessage(withImage); + const after = convertToLlm(messages); + const afterUser = after.find(entry => entry.role === "user"); + expect(Array.isArray(afterUser?.content) && afterUser.content.some(b => b.type === "image")).toBe(false); + }); +}); + describe("replaceLlmImagesWithText", () => { it("replaces image blocks in user, developer, and tool-result messages with the placeholder", () => { const converted = convertToLlm([ diff --git a/packages/coding-agent/src/session/messages.ts b/packages/coding-agent/src/session/messages.ts index 6fd7b24ed..97a607ddf 100644 --- a/packages/coding-agent/src/session/messages.ts +++ b/packages/coding-agent/src/session/messages.ts @@ -5,6 +5,10 @@ * and provides a transformer to convert them to LLM-compatible messages. */ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { + invalidateMessageCache, + registerMessageCacheInvalidator, +} from "@oh-my-pi/pi-agent-core/compaction/message-cache"; import { type BranchSummaryMessage, type CompactionSummaryMessage, @@ -457,6 +461,14 @@ function stripImagesFromArrayContent(content: (TextContent | ImageContent)[]): S * pure local mutation and intentionally does neither. */ export function stripImagesFromMessage(message: AgentMessage): number { + const removed = stripImagesFromMessageContent(message); + // The mutated message keeps its identity across context rebuilds, so drop its + // cached estimate/convert before the next pass counts/converts the new shape. + if (removed > 0) invalidateMessageCache(message); + return removed; +} + +function stripImagesFromMessageContent(message: AgentMessage): number { switch (message.role) { case "user": case "developer": @@ -750,6 +762,184 @@ function convertImageBearingCustomMessage(message: CustomMessage | HookMessage): return converted; } +/** + * Per-message conversion result, keyed by message identity. `interruptedNext` + * records the neighbor state the fragment was built against so an assistant + * whose following {@link INTERRUPTED_THINKING_MESSAGE_TYPE} marker appears or + * disappears is recomputed (its LLM view strips the trailing thinking run only + * while that marker follows). + * + * WeakMap (not a symbol tag) is deliberate: `wrapSteeringForModel` and + * `deobfuscateAgentMessages` spread messages into fresh variants with different + * content; a symbol-keyed fragment would ride that spread and mis-convert the + * copy. Identity keying keeps the cache off spread copies. + */ +interface ConvertMemoEntry { + interruptedNext: boolean; + fragment: Message[]; +} +const convertCache = new WeakMap(); + +// Array-level shortcuts over the per-message memo. The live agent mutates one +// `AgentMessage[]` identity across a turn: appending new messages and swapping +// the streaming tail (`context.messages[len-1] = partial → trailing`). Between +// owner invalidations (prune/shake/strip bump `convertGeneration`) and for a +// given array identity, only the last index is ever swapped and the array only +// grows — interior prefix messages are immutable. That invariant lets two +// shortcuts skip the O(N) re-walk: +// - exact-repeat: same array, same length, same generation, same tail identity +// → hand back the same outer array. +// - slice-on-growth: same array, same generation, length grew → copy the +// unchanged prefix output and reconvert only the neighbor-sensitive boundary +// message plus the appended suffix. +// The tail-identity guard on exact-repeat catches the streaming snapshot swap +// (partial → trailing is a fresh identity), so a settled tail is never served +// from a stale mid-stream fragment. +let convertGeneration = 0; +let lastConvertInput: AgentMessage[] | undefined; +let lastConvertLength = 0; +let lastConvertOutput: Message[] | undefined; +let lastConvertGeneration = -1; +let lastConvertTail: AgentMessage | undefined; +// Output-message count contributed by messages[0 .. lastConvertLength-1), i.e. +// every message except the last. The last message is neighbor-sensitive (its LLM +// view drops the trailing thinking run only while an interrupted-thinking marker +// follows), so growth reconverts it rather than reusing its old fragment. +let lastConvertPrefixOutputLen = 0; + +registerMessageCacheInvalidator(message => { + convertCache.delete(message); + convertGeneration++; +}); + +/** Convert one message to its LLM fragment. `interruptedNext` is true only for an + * assistant turn immediately followed by its interrupted-thinking marker. */ +function convertOne(m: AgentMessage, interruptedNext: boolean): Message[] { + switch (m.role) { + case "bashExecution": + if (m.excludeFromContext) { + return []; + } + return [ + { + role: "user", + content: [{ type: "text", text: bashExecutionToText(m) }], + attribution: "user", + timestamp: m.timestamp, + }, + ]; + case "pythonExecution": + if (m.excludeFromContext) { + return []; + } + return [ + { + role: "user", + content: [{ type: "text", text: pythonExecutionToText(m) }], + attribution: "user", + timestamp: m.timestamp, + }, + ]; + case "fileMention": { + // One `fileMention` can mix `@notes.md` (text) and `@screenshot.png` (image) + // in the same turn (`generateFileMentionMessages` packs every `@…` into a + // single message). Splitting by image presence keeps text-only mentions on + // the higher-priority `developer` slot while routing image attachments + // through `user`, the only Responses content slot that legitimately accepts + // `input_image` (Codex chatgpt.com /codex/responses rejects everything else + // with `Invalid value: 'input_image'`, #3443). + const wrap = (file: FileMentionMessage["files"][number]): string => { + const inner = file.content ? `\n${file.content}\n` : "\n"; + return `${inner}`; + }; + const textFiles = m.files.filter(file => !file.image); + const imageFiles = m.files.filter(file => file.image); + const out: Message[] = []; + if (textFiles.length > 0) { + out.push({ + role: "developer", + content: [{ type: "text" as const, text: textFiles.map(wrap).join("\n") }], + attribution: "user", + timestamp: m.timestamp, + }); + } + if (imageFiles.length > 0) { + const content: (TextContent | ImageContent)[] = [ + { type: "text" as const, text: imageFiles.map(wrap).join("\n") }, + ]; + for (const file of imageFiles) { + if (file.image) content.push(file.image); + } + out.push({ + role: "user", + content, + attribution: "user", + timestamp: m.timestamp, + }); + } + return out; + } + case "custom": { + if (!isCustomMessageContent(m.content)) return []; + if (isUserInvokedSkillPrompt(m)) { + return [ + { + role: "user", + content: customMessageContentToLlmContent(m.content), + attribution: "user", + timestamp: m.timestamp, + }, + ]; + } + const split = convertImageBearingCustomMessage(m); + if (split) return split; + const converted = convertMessageToLlm(m); + return converted ? [converted] : []; + } + case "hookMessage": { + if (!isCustomMessageContent(m.content)) return []; + const split = convertImageBearingCustomMessage(m); + if (split) return split; + const converted = convertMessageToLlm(m); + return converted ? [converted] : []; + } + case "assistant": { + // A user-interrupted turn keeps its trailing thinking run on the + // persisted/displayed message so reload and Ctrl+L rebuilds still + // show it. That run is incomplete/unsigned and gets rejected on + // resend, so strip it here — LLM path only — when the hidden + // interrupted-thinking continuity message follows. + const source = interruptedNext ? stripDemotedThinkingForLlm(m) : m; + const converted = convertMessageToLlm(source); + return converted ? [converted] : []; + } + case "branchSummary": + case "compactionSummary": + case "user": + case "developer": + case "toolResult": { + // Core roles share one transformer with agent-core — + // duplicating them here is how snapcompact frames once + // silently fell off the provider request. + const converted = convertMessageToLlm(m); + return converted ? [converted] : []; + } + default: + m satisfies never; + return []; + } +} + +/** Cached per-message conversion. Reuses the stored fragment while identity and + * `interruptedNext` neighbor state hold; recomputes on a neighbor flip. */ +function convertOneCached(m: AgentMessage, interruptedNext: boolean): Message[] { + const cached = convertCache.get(m); + if (cached !== undefined && cached.interruptedNext === interruptedNext) return cached.fragment; + const fragment = convertOne(m, interruptedNext); + convertCache.set(m, { interruptedNext, fragment }); + return fragment; +} + /** * Transform AgentMessages (including custom types) to LLM-compatible Messages. * @@ -757,121 +947,69 @@ function convertImageBearingCustomMessage(message: CustomMessage | HookMessage): * - Agent's transormToLlm option (for prompt calls and queued messages) * - Compaction's generateSummary (for summarization) * - Custom extensions and tools + * + * Settled history converts once and is reused per message identity: an + * append-only turn on the same array re-pays only the new suffix, and an + * unchanged re-convert of the same array hands back the same outer `Message[]`. + * Owner mutations (prune/shake/strip-images) invalidate the affected message + * through the shared registry before the next pass. */ export function convertToLlm(messages: AgentMessage[]): Message[] { - return messages.flatMap((m, index): Message[] => { - switch (m.role) { - case "bashExecution": - if (m.excludeFromContext) { - return []; - } - return [ - { - role: "user", - content: [{ type: "text", text: bashExecutionToText(m) }], - attribution: "user", - timestamp: m.timestamp, - }, - ]; - case "pythonExecution": - if (m.excludeFromContext) { - return []; - } - return [ - { - role: "user", - content: [{ type: "text", text: pythonExecutionToText(m) }], - attribution: "user", - timestamp: m.timestamp, - }, - ]; - case "fileMention": { - // One `fileMention` can mix `@notes.md` (text) and `@screenshot.png` (image) - // in the same turn (`generateFileMentionMessages` packs every `@…` into a - // single message). Splitting by image presence keeps text-only mentions on - // the higher-priority `developer` slot while routing image attachments - // through `user`, the only Responses content slot that legitimately accepts - // `input_image` (Codex chatgpt.com /codex/responses rejects everything else - // with `Invalid value: 'input_image'`, #3443). - const wrap = (file: FileMentionMessage["files"][number]): string => { - const inner = file.content ? `\n${file.content}\n` : "\n"; - return `${inner}`; - }; - const textFiles = m.files.filter(file => !file.image); - const imageFiles = m.files.filter(file => file.image); - const out: Message[] = []; - if (textFiles.length > 0) { - out.push({ - role: "developer", - content: [{ type: "text" as const, text: textFiles.map(wrap).join("\n") }], - attribution: "user", - timestamp: m.timestamp, - }); - } - if (imageFiles.length > 0) { - const content: (TextContent | ImageContent)[] = [ - { type: "text" as const, text: imageFiles.map(wrap).join("\n") }, - ]; - for (const file of imageFiles) { - if (file.image) content.push(file.image); - } - out.push({ - role: "user", - content, - attribution: "user", - timestamp: m.timestamp, - }); - } - return out; - } - case "custom": { - if (!isCustomMessageContent(m.content)) return []; - if (isUserInvokedSkillPrompt(m)) { - return [ - { - role: "user", - content: customMessageContentToLlmContent(m.content), - attribution: "user", - timestamp: m.timestamp, - }, - ]; - } - const split = convertImageBearingCustomMessage(m); - if (split) return split; - const converted = convertMessageToLlm(m); - return converted ? [converted] : []; - } - case "hookMessage": { - if (!isCustomMessageContent(m.content)) return []; - const split = convertImageBearingCustomMessage(m); - if (split) return split; - const converted = convertMessageToLlm(m); - return converted ? [converted] : []; - } - case "assistant": { - // A user-interrupted turn keeps its trailing thinking run on the - // persisted/displayed message so reload and Ctrl+L rebuilds still - // show it. That run is incomplete/unsigned and gets rejected on - // resend, so strip it here — LLM path only — when the hidden - // interrupted-thinking continuity message follows. - const source = followedByInterruptedThinking(messages, index) ? stripDemotedThinkingForLlm(m) : m; - const converted = convertMessageToLlm(source); - return converted ? [converted] : []; - } - case "branchSummary": - case "compactionSummary": - case "user": - case "developer": - case "toolResult": { - // Core roles share one transformer with agent-core — - // duplicating them here is how snapcompact frames once - // silently fell off the provider request. - const converted = convertMessageToLlm(m); - return converted ? [converted] : []; - } - default: - m satisfies never; - return []; - } - }); + const len = messages.length; + const sameArray = messages === lastConvertInput && lastConvertGeneration === convertGeneration; + const tail = len > 0 ? messages[len - 1] : undefined; + + // Exact-repeat: same array, same length, same trailing identity → reuse the + // outer array. The tail-identity check rejects the streaming snapshot swap + // (partial → settled trailing keeps array identity/length but mints a fresh + // tail), so a settled tail never reads a stale mid-stream fragment. + if (sameArray && lastConvertOutput !== undefined && len === lastConvertLength && tail === lastConvertTail) { + return lastConvertOutput; + } + + // Slice-on-growth: same array grew by append. Every interior message is + // immutable under one array identity, so copy the unchanged prefix output + // (messages[0 .. lastLen-1)) and reconvert only the old boundary message + // (neighbor-sensitive: a following interrupted-thinking marker may now exist) + // plus the appended suffix. The boundary-identity check (old tail still sits + // at its old index) rejects an in-place interior splice-replace that grew the + // array while swapping earlier identities, forcing a full rebuild. + let out: Message[]; + let start: number; + if ( + sameArray && + lastConvertOutput !== undefined && + len > lastConvertLength && + lastConvertLength > 0 && + messages[lastConvertLength - 1] === lastConvertTail && + lastConvertPrefixOutputLen <= lastConvertOutput.length + ) { + out = lastConvertOutput.slice(0, lastConvertPrefixOutputLen); + start = lastConvertLength - 1; + } else { + out = []; + start = 0; + } + + // Output length contributed by messages[0 .. len-1), captured when the loop + // reaches the final index so the next growth can reuse this prefix. + let prefixOutputLen = 0; + for (let i = start; i < len; i++) { + if (i === len - 1) prefixOutputLen = out.length; + const m = messages[i]; + const interruptedNext = m.role === "assistant" && followedByInterruptedThinking(messages, i); + const fragment = convertOneCached(m, interruptedNext); + for (const msg of fragment) out.push(msg); + } + if (len === 0) prefixOutputLen = 0; + + // Record for the next call's shortcuts. `out` is a fresh array (slice or new), + // so a prior caller holding the previous `lastConvertOutput` never sees it grow. + lastConvertInput = messages; + lastConvertLength = len; + lastConvertOutput = out; + lastConvertGeneration = convertGeneration; + lastConvertTail = tail; + lastConvertPrefixOutputLen = prefixOutputLen; + return out; } diff --git a/packages/coding-agent/src/session/session-context.test.ts b/packages/coding-agent/src/session/session-context.test.ts index 914434999..d03d6c1c2 100644 --- a/packages/coding-agent/src/session/session-context.test.ts +++ b/packages/coding-agent/src/session/session-context.test.ts @@ -1,7 +1,8 @@ import { describe, expect, it } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; import * as snapcompact from "@oh-my-pi/snapcompact"; -import type { CompactionSummaryMessage } from "./messages"; +import { type CompactionSummaryMessage, INTERRUPTED_THINKING_MESSAGE_TYPE } from "./messages"; import { buildSessionContext, type StrippedToolCallsMarker } from "./session-context"; import type { SessionEntry } from "./session-entries"; @@ -159,3 +160,225 @@ describe("buildSessionContext dangling toolCalls", () => { expect(context.messages.some(message => message.role === "assistant")).toBe(false); }); }); + +const assistantUsage: AssistantMessage["usage"] = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function userEntry(id: string, parentId: string | null, content: string, messageTimestamp: number): SessionEntry { + return { + type: "message", + id, + parentId, + timestamp, + message: { role: "user", content, timestamp: messageTimestamp } as AgentMessage, + }; +} + +function assistantEntry( + id: string, + parentId: string | null, + stopReason: AssistantMessage["stopReason"], + text: string, + messageTimestamp: number, +): SessionEntry { + return { + type: "message", + id, + parentId, + timestamp, + message: { + role: "assistant", + content: [{ type: "text", text }], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-sonnet-4-5", + usage: assistantUsage, + stopReason, + timestamp: messageTimestamp, + } satisfies AssistantMessage, + }; +} + +function toolCallAssistantEntry( + id: string, + parentId: string | null, + stopReason: AssistantMessage["stopReason"], + toolCallId: string, + messageTimestamp: number, +): SessionEntry { + return { + type: "message", + id, + parentId, + timestamp, + message: { + role: "assistant", + content: [{ type: "toolCall", id: toolCallId, name: "write", arguments: { path: "plan.md", content: "x" } }], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-sonnet-4-5", + usage: assistantUsage, + stopReason, + timestamp: messageTimestamp, + } satisfies AssistantMessage, + }; +} + +function syntheticToolResultEntry( + id: string, + parentId: string | null, + toolCallId: string, + messageTimestamp: number, +): SessionEntry { + return { + type: "message", + id, + parentId, + timestamp, + message: { + role: "toolResult", + toolCallId, + toolName: "write", + content: [ + { type: "text", text: "Tool call was not executed because the provider stream ended with an error." }, + ], + details: { __synthetic: true, source: "assistant_stop_error", executed: false }, + isError: true, + timestamp: messageTimestamp, + } as AgentMessage, + }; +} + +function hiddenContinuityEntry(id: string, parentId: string | null): SessionEntry { + return { + type: "custom_message", + id, + parentId, + timestamp, + customType: INTERRUPTED_THINKING_MESSAGE_TYPE, + content: "preserved interrupted thinking", + display: false, + attribution: "agent", + }; +} + +function expectUserTail(messages: AgentMessage[], content: string): void { + const tail = messages.at(-1); + expect(tail?.role).toBe("user"); + if (tail?.role !== "user") { + throw new Error(`Expected user tail, received ${tail?.role ?? "none"}`); + } + expect(tail.content).toBe(content); +} + +describe("buildSessionContext failed replay tails", () => { + it("terminates on cyclic parent links and includes each reachable message once", () => { + const entries = [userEntry("A", "B", "from A", 1), userEntry("B", "A", "from B", 2)]; + + const context = buildSessionContext(entries, "A"); + + expect(context.messages.map(message => (message.role === "user" ? message.content : message.role))).toEqual([ + "from B", + "from A", + ]); + }); + + it("omits a terminal aborted assistant from normal context", () => { + const context = buildSessionContext([ + userEntry("user", null, "continue", 1), + assistantEntry("assistant", "user", "aborted", "partial unsafe replay", 2), + ]); + + expect(context.messages.some(message => message.role === "assistant")).toBe(false); + expectUserTail(context.messages, "continue"); + }); + + it("omits an earlier aborted assistant before a later user from normal context", () => { + const context = buildSessionContext([ + userEntry("user-1", null, "first prompt", 1), + assistantEntry("assistant", "user-1", "aborted", "partial unsafe replay", 2), + userEntry("user-2", "assistant", "retry", 3), + ]); + + expect(context.messages.some(message => message.role === "assistant")).toBe(false); + expectUserTail(context.messages, "retry"); + }); + + it("preserves a terminal aborted assistant in transcript mode", () => { + const context = buildSessionContext( + [ + userEntry("user", null, "continue", 1), + assistantEntry("assistant", "user", "aborted", "visible transcript error", 2), + ], + undefined, + undefined, + { transcript: true }, + ); + + const assistant = context.messages.find(message => message.role === "assistant"); + expect(assistant?.role).toBe("assistant"); + if (assistant?.role !== "assistant") { + throw new Error(`Expected transcript assistant, received ${assistant?.role ?? "none"}`); + } + expect(assistant.stopReason).toBe("aborted"); + expect(assistant.content).toEqual([{ type: "text", text: "visible transcript error" }]); + }); + + it("omits a terminal error assistant from normal context", () => { + const context = buildSessionContext([ + userEntry("user", null, "retry with smaller input", 1), + assistantEntry("assistant", "user", "error", "provider rejected the request", 2), + ]); + + expect(context.messages.some(message => message.role === "assistant")).toBe(false); + expectUserTail(context.messages, "retry with smaller input"); + }); + + it("keeps an aborted assistant when hidden interrupted-thinking continuity follows it", () => { + const context = buildSessionContext([ + userEntry("user", null, "keep reasoning continuity", 1), + assistantEntry("assistant", "user", "aborted", "partial answer before interrupt", 2), + hiddenContinuityEntry("continuity", "assistant"), + ]); + + const assistant = context.messages.find(message => message.role === "assistant"); + expect(assistant?.role).toBe("assistant"); + if (assistant?.role !== "assistant") { + throw new Error(`Expected assistant before continuity, received ${assistant?.role ?? "none"}`); + } + expect(assistant.stopReason).toBe("aborted"); + expect(context.messages.at(-1)?.role).toBe("custom"); + }); + + it("drops synthetic tool results paired with a dropped failed tool-call turn", () => { + const context = buildSessionContext([ + userEntry("user", null, "write the plan", 1), + toolCallAssistantEntry("assistant", "user", "error", "call-1", 2), + syntheticToolResultEntry("result", "assistant", "call-1", 3), + ]); + + expect(context.messages.map(message => message.role)).toEqual(["user"]); + expectUserTail(context.messages, "write the plan"); + }); + + it("keeps the failed tool-call turn and its result in transcript mode", () => { + const context = buildSessionContext( + [ + userEntry("user", null, "write the plan", 1), + toolCallAssistantEntry("assistant", "user", "error", "call-1", 2), + syntheticToolResultEntry("result", "assistant", "call-1", 3), + ], + undefined, + undefined, + { transcript: true, keepDanglingToolCalls: true }, + ); + + expect(context.messages.map(message => message.role)).toEqual(["user", "assistant", "toolResult"]); + }); +}); diff --git a/packages/coding-agent/src/session/session-context.ts b/packages/coding-agent/src/session/session-context.ts index 599963055..abe1502e6 100644 --- a/packages/coding-agent/src/session/session-context.ts +++ b/packages/coding-agent/src/session/session-context.ts @@ -5,6 +5,7 @@ import { createBranchSummaryMessage, createCompactionSummaryMessage, createCustomMessage, + INTERRUPTED_THINKING_MESSAGE_TYPE, isCustomMessageContent, normalizeCustomMessagePayload, } from "./messages"; @@ -197,10 +198,13 @@ export function buildSessionContext( }; } - // Walk from leaf to root, collecting path + // Walk from leaf to root, collecting path. Corrupt/pre-fix files can contain + // parent cycles; stop at the first repeat so session load is bounded. const path: SessionEntry[] = []; + const seenPathIds = new Set(); let current: SessionEntry | undefined = leaf; - while (current) { + while (current && !seenPathIds.has(current.id)) { + seenPathIds.add(current.id); path.push(current); current = current.parentId ? byId.get(current.parentId) : undefined; } @@ -498,6 +502,41 @@ export function buildSessionContext( } } + // Error/abort assistant turns are transcript events, not safe assistant + // turns to replay into the next provider request. Drop them even when a + // later user message follows through non-context entries (`session_exit`, + // labels, etc.); otherwise a resumed session replays a dead partial turn + // and can spend minutes reprocessing old context before the new prompt. + // Keep the interrupted-thinking continuity pair: convertToLlm strips the + // unsafe trailing thinking from that assistant and sends the hidden + // continuity note instead. + if (!options?.transcript) { + for (let i = messages.length - 1; i >= 0; i--) { + const message = messages[i]; + if (message?.role !== "assistant") continue; + if (message.stopReason !== "aborted" && message.stopReason !== "error") continue; + const next = messages[i + 1]; + if (next?.role === "custom" && next.customType === INTERRUPTED_THINKING_MESSAGE_TYPE) continue; + // A failed turn that emitted tool calls persists paired synthetic + // tool_result placeholders after it. Dropping only the assistant would + // strand those results with no preceding tool_use — a shape providers + // reject — so remove the paired results alongside the turn. + const droppedToolCallIds = new Set(); + for (const block of message.content) { + if (block.type === "toolCall") droppedToolCallIds.add(block.id); + } + messages.splice(i, 1); + if (droppedToolCallIds.size > 0) { + for (let j = messages.length - 1; j >= i; j--) { + const candidate = messages[j]; + if (candidate?.role === "toolResult" && droppedToolCallIds.has(candidate.toolCallId)) { + messages.splice(j, 1); + } + } + } + } + } + return { messages, cacheMissExplainedAt: options?.transcript ? cacheMissExplainedAt : undefined, diff --git a/packages/coding-agent/src/session/session-entries.ts b/packages/coding-agent/src/session/session-entries.ts index a6b6ee3cd..54caecbf7 100644 --- a/packages/coding-agent/src/session/session-entries.ts +++ b/packages/coding-agent/src/session/session-entries.ts @@ -1,5 +1,6 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai"; +import type { StructuredSubagentSchemaMode } from "../task/types"; export const CURRENT_SESSION_VERSION = 3; @@ -164,8 +165,12 @@ export interface SessionInitEntry extends SessionEntryBase { task: string; /** Tools available to the agent */ tools: string[]; - /** Output schema if structured output was requested */ + /** Output schema if structured output was requested. */ outputSchema?: unknown; + /** Enforcement policy recorded with the output schema for faithful revival. */ + outputSchemaMode?: StructuredSubagentSchemaMode; + /** Whether revival must retain only the explicitly persisted tool names. */ + restrictToolNames?: boolean; /** Spawn allowlist the subagent ran with ("" = none, "*" = any, else CSV); absent on pre-spawns files. */ spawns?: string; /** The agent's `readSummarize` setting (`false` = read summarization disabled); absent uses the session default. */ diff --git a/packages/coding-agent/src/session/session-history-format.ts b/packages/coding-agent/src/session/session-history-format.ts index de4d07041..a88145d3e 100644 --- a/packages/coding-agent/src/session/session-history-format.ts +++ b/packages/coding-agent/src/session/session-history-format.ts @@ -46,6 +46,15 @@ export interface HistoryFormatOptions { * this so it sees what changed without re-reading the file. */ expandEditDiffs?: boolean; + /** + * Chunked rendering support: a caller formatting one logical transcript in + * several calls (the advisor's chunked delta render) passes a result index + * built over the WHOLE delta plus one shared consumed-id set, so a toolCall + * finds its toolResult across chunk boundaries and the result is never + * re-rendered as an orphan in a later chunk. + */ + toolResultIndex?: ReadonlyMap; + consumedToolCallIds?: Set; } /** Max length of the primary-arg summary inside `→ tool(...)` lines. */ @@ -273,13 +282,19 @@ export function formatSessionHistoryMarkdown(messages: unknown[], opts?: History } // Index tool results by call id so each toolCall collapses to one line. - const resultsByCallId = new Map(); - for (const msg of typed) { - if (msg.role === "toolResult") { - resultsByCallId.set(msg.toolCallId, msg); + // Chunked callers supply a whole-delta index + shared consumed set so + // call/result pairs resolve across chunk boundaries. + let resultsByCallId = opts?.toolResultIndex; + if (!resultsByCallId) { + const local = new Map(); + for (const msg of typed) { + if (msg.role === "toolResult") { + local.set(msg.toolCallId, msg); + } } + resultsByCallId = local; } - const consumed = new Set(); + const consumed = opts?.consumedToolCallIds ?? new Set(); // In watched mode, consecutive same-role messages collapse under one label // (the watched agent emits one assistant message per tool call, so otherwise // every call repeats `**agent**:`). Cleared whenever a diff --git a/packages/coding-agent/src/session/session-listing.ts b/packages/coding-agent/src/session/session-listing.ts index 4b554fddb..586cbd825 100644 --- a/packages/coding-agent/src/session/session-listing.ts +++ b/packages/coding-agent/src/session/session-listing.ts @@ -1,6 +1,6 @@ import * as os from "node:os"; import * as path from "node:path"; -import type { Message, TextContent } from "@oh-my-pi/pi-ai"; +import type { Message } from "@oh-my-pi/pi-ai"; import { getAgentDir as getDefaultAgentDir, logger, parseJsonlLenient, toError } from "@oh-my-pi/pi-utils"; import { computeDefaultSessionDir } from "./session-paths"; import { FileSessionStorage, type SessionStorage } from "./session-storage"; @@ -108,10 +108,11 @@ function sessionDisplayName(info: SessionInfo): string { function extractTextFromContent(content: Message["content"]): string { if (typeof content === "string") return content; - return content - .filter((block): block is TextContent => block.type === "text") - .map(block => block.text) - .join(" "); + const text: string[] = []; + for (const block of content) { + if (block.type === "text") text.push(block.text); + } + return text.join(" "); } /** diff --git a/packages/coding-agent/src/session/session-loader.ts b/packages/coding-agent/src/session/session-loader.ts index 0915db7b5..05db3fdb5 100644 --- a/packages/coding-agent/src/session/session-loader.ts +++ b/packages/coding-agent/src/session/session-loader.ts @@ -252,6 +252,15 @@ async function resolvePersistedBlobRefs(value: unknown, blobStore: BlobStore, ke } if (typeof value !== "object" || value === null) return; + if ( + "type" in value && + value.type === "image_generation_call" && + "result" in value && + typeof value.result === "string" && + isBlobRef(value.result) + ) { + value.result = await resolveImageData(blobStore, value.result); + } if (hasImageUrl(value) && isBlobRef(value.image_url)) { value.image_url = await resolveImageDataUrl(blobStore, value.image_url); @@ -262,17 +271,47 @@ async function resolvePersistedBlobRefs(value: unknown, blobStore: BlobStore, ke ); } +/** + * Cheap synchronous precheck: does this value's tree contain any `blob:sha256:` string? + * Early-exits on the first hit and allocates no promises, so blob-free entries skip the + * async {@link resolvePersistedBlobRefs} descent entirely. Conservative — a blob ref in a + * non-resolved position still returns true, which only costs an extra (no-op) walk. + */ +function containsBlobRef(value: unknown): boolean { + if (typeof value === "string") return isBlobRef(value); + if (Array.isArray(value)) { + for (const item of value) { + if (containsBlobRef(item)) return true; + } + return false; + } + if (typeof value !== "object" || value === null) return false; + for (const key in value) { + if (containsBlobRef((value as Record)[key])) return true; + } + return false; +} + export async function resolveBlobRefsInEntries(entries: FileEntry[], blobStore: BlobStore): Promise { - await Promise.all( - entries.filter(entry => entry.type !== "session").map(entry => resolvePersistedBlobRefs(entry, blobStore)), - ); + const pending: Promise[] = []; + // Interleave precheck + initiation per entry so a positive entry begins resolution at the same + // relative point as the old filter+map schedule (no scan-all-first pass that could observe a + // later entry before an earlier resolution mutates it). + for (const entry of entries) { + if (entry.type === "session") continue; + if (!containsBlobRef(entry)) continue; + pending.push(resolvePersistedBlobRefs(entry, blobStore)); + } + await Promise.all(pending); } /** - * Read-only message view of a session file: load entries, migrate to the - * current version, resolve blob refs, and build the context along the - * persisted leaf path (last entry). Does NOT create a writer or take the - * session lock — safe to call against a file another session is writing. + * Read-only transcript view of a session file: load entries, migrate to the + * current version, resolve blob refs, and build the display transcript along + * the persisted leaf path (last entry). Uses transcript mode (collapsed to the + * latest compaction) so failed/aborted tail turns stay visible, unlike the + * provider-context builder which drops them. Does NOT create a writer or take + * the session lock — safe to call against a file another session is writing. */ export async function loadSessionMessagesReadOnly(filePath: string): Promise { const entries = await loadEntriesFromFile(filePath); @@ -280,5 +319,8 @@ export async function loadSessionMessagesReadOnly(filePath: string): Promise e.type !== "session"); - return buildSessionContext(sessionEntries).messages; + return buildSessionContext(sessionEntries, undefined, undefined, { + transcript: true, + collapseCompactedHistory: true, + }).messages; } diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index b259dd8c9..ef78771d4 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -18,6 +18,7 @@ import { stringifyJson, toError, } from "@oh-my-pi/pi-utils"; +import type { StructuredSubagentSchemaMode } from "../task/types"; import { ArtifactManager } from "./artifacts"; import { type BlobPutOptions, type BlobPutResult, BlobStore } from "./blob-store"; import { @@ -445,6 +446,13 @@ export class SessionManager { #inMemoryArtifactCounter = 0; #suppressBreadcrumb = false; + /** + * The last breadcrumb this manager wrote marked a lazy `/new` boundary whose + * JSONL is not yet on disk. Cleared (and the crumb re-stamped non-fresh) once + * the session materializes, so a materialized-then-deleted session still falls + * back to the most-recent session instead of being treated as a fresh crumb. + */ + #breadcrumbFresh = false; #sessionNameChangedCallbacks = new Set<() => void>(); private constructor(cwd: string, sessionDir: string, persist: boolean, storage: SessionStorage) { @@ -457,8 +465,18 @@ export class SessionManager { if (persist && sessionDir) this.#storage.ensureDirSync(sessionDir); } - #rememberBreadcrumb(cwd: string, sessionFile: string): void { - if (!this.#suppressBreadcrumb) writeTerminalBreadcrumb(cwd, sessionFile); + #rememberBreadcrumb(cwd: string, sessionFile: string, fresh = false): void { + this.#breadcrumbFresh = fresh; + if (!this.#suppressBreadcrumb) writeTerminalBreadcrumb(cwd, sessionFile, fresh); + } + + /** + * Re-stamp a fresh `/new` breadcrumb as non-fresh once the session has + * materialized on disk. A no-op unless the current breadcrumb is still fresh. + */ + #materializeBreadcrumb(): void { + if (!this.#breadcrumbFresh || !this.#sessionFile) return; + this.#rememberBreadcrumb(this.#cwd, this.#sessionFile, false); } #clearDiskError(): void { @@ -600,6 +618,7 @@ export class SessionManager { this.#closeWriterEventually(); this.#storage.writeTextSync(this.#sessionFile, body); this.#fileIsCurrent = true; + this.#materializeBreadcrumb(); this.#rewriteRequired = false; this.#hasTitleSlot = true; } catch (err) { @@ -625,6 +644,7 @@ export class SessionManager { async () => { if (await this.#runFencedAtomicRewrite(startEpoch)) { this.#fileIsCurrent = true; + this.#materializeBreadcrumb(); this.#rewriteRequired = false; this.#hasTitleSlot = true; } @@ -807,7 +827,7 @@ export class SessionManager { this.#sessionFile = forcedSessionFile ?? path.join(this.#sessionDir, `${fileSafeTimestamp(timestamp)}_${this.#sessionId}.jsonl`); - this.#rememberBreadcrumb(this.#cwd, this.#sessionFile); + this.#rememberBreadcrumb(this.#cwd, this.#sessionFile, true); } else { this.#sessionFile = undefined; } @@ -935,6 +955,26 @@ export class SessionManager { }; } + /** + * Create an independent manager for the current logical session and branch. + * The clone shares the storage backend but owns its entry index and writer, so + * callers can finish session-owned work after this manager switches elsewhere. + * Set `persist` false when the original session is intentionally being dropped. + */ + cloneCurrentSession(options?: { persist?: boolean }): SessionManager { + const persist = options?.persist ?? this.#persist; + const clone = new SessionManager(this.#cwd, this.#sessionDir, persist, this.#storage); + clone.#suppressBreadcrumb = true; + clone.restoreState(this.captureState()); + if (!persist) { + clone.#sessionFile = undefined; + clone.#fileIsCurrent = false; + clone.#rewriteRequired = false; + clone.#forceFileCreation = false; + } + return clone; + } + restoreState(snapshot: SessionManagerStateSnapshot): void { this.#closeWriterEventually(); this.#diskTail = Promise.resolve(); @@ -1470,6 +1510,34 @@ export class SessionManager { return entry.id; } + /** + * Append to a non-active branch without changing the current leaf. + * Used by work that retains ownership of a branch across tree navigation. + */ + appendMessageToBranch( + message: + | Message + | CustomMessage + | HookMessage + | BashExecutionMessage + | PythonExecutionMessage + | FileMentionMessage, + parentId: string | null, + ): string { + if (parentId !== null && !this.#index.has(parentId)) throw new Error(`Entry ${parentId} not found`); + const activeLeafId = this.#index.leafId(); + const entry: SessionMessageEntry = { + type: "message", + id: generateId(this.#index), + parentId, + timestamp: nowIso(), + message, + }; + this.#recordEntry(entry); + this.#index.setLeaf(activeLeafId); + return entry.id; + } + /** Append a thinking level change as child of current leaf, then advance leaf. Returns entry id. */ appendThinkingLevelChange(thinkingLevel?: string, configured?: string): string { const entry: ThinkingLevelChangeEntry = { @@ -1510,6 +1578,8 @@ export class SessionManager { task: string; tools: string[]; outputSchema?: unknown; + outputSchemaMode?: StructuredSubagentSchemaMode; + restrictToolNames?: boolean; spawns?: string; readSummarize?: boolean; }): string { @@ -1948,6 +2018,8 @@ export class SessionManager { task: string; tools: string[]; outputSchema?: unknown; + outputSchemaMode?: StructuredSubagentSchemaMode; + restrictToolNames?: boolean; spawns?: string; readSummarize?: boolean; } | null; @@ -1966,6 +2038,8 @@ export class SessionManager { task: string; tools: string[]; outputSchema?: unknown; + outputSchemaMode?: StructuredSubagentSchemaMode; + restrictToolNames?: boolean; spawns?: string; readSummarize?: boolean; } | null = null; @@ -1977,6 +2051,8 @@ export class SessionManager { task: entry.task, tools: entry.tools, outputSchema: entry.outputSchema, + outputSchemaMode: entry.outputSchemaMode, + restrictToolNames: entry.restrictToolNames, readSummarize: entry.readSummarize, spawns: entry.spawns, }; @@ -1998,6 +2074,18 @@ export class SessionManager { let chosenSession: string | null | undefined; if (breadcrumb) { + // A fresh `/new` boundary whose JSONL was never materialized (lazy + // new-session persistence, then a process exit before any assistant + // output). Honor the boundary: start fresh rather than falling back to + // findMostRecentSession(), which would resurrect the pre-`/new` + // transcript. A materialized (or genuinely stale/deleted) crumb reports + // exists=false only when fresh, so this never masks a real stale crumb. + if (breadcrumb.fresh && !breadcrumb.exists) { + const manager = new SessionManager(cwd, dir, true, storage); + manager.#resetToNewSession(); + return manager; + } + // Recover stale crumbs: a subagent open (pre-fix) may have pointed this // terminal's breadcrumb at an artifact child; resume the parent instead. breadcrumb.sessionFile = resolveBreadcrumbToInteractiveRoot(breadcrumb.sessionFile); diff --git a/packages/coding-agent/src/session/session-paths.ts b/packages/coding-agent/src/session/session-paths.ts index 79871f800..237aa013e 100644 --- a/packages/coding-agent/src/session/session-paths.ts +++ b/packages/coding-agent/src/session/session-paths.ts @@ -146,28 +146,54 @@ export function computeDefaultSessionDir( * Write a breadcrumb linking the current terminal to a session file. * The breadcrumb contains the cwd and session path so --continue can * find "this terminal's last session" even when running concurrent instances. + * + * `fresh` marks a `/new` (or freshly-minted) session boundary whose JSONL is + * not yet materialized (new-session persistence is lazy until assistant output + * exists). A fresh breadcrumb is honored by {@link readTerminalBreadcrumbEntry} + * even when its target file is still absent, so relaunch/auto-resume reopens the + * post-`/new` session instead of falling back to the pre-`/new` transcript. Once + * the session materializes the caller rewrites the breadcrumb with `fresh:false` + * so a later external delete is still treated as a genuinely stale crumb. */ -export function writeTerminalBreadcrumb(cwd: string, sessionFile: string): void { +export function writeTerminalBreadcrumb(cwd: string, sessionFile: string, fresh = false): void { const terminalId = getTerminalId(); if (!terminalId) return; const breadcrumbDir = getTerminalSessionsDir(); const breadcrumbFile = path.join(breadcrumbDir, terminalId); - const content = `${cwd}\n${sessionFile}\n`; - // Best-effort — don't break session creation if breadcrumb fails - Bun.write(breadcrumbFile, content).catch(() => {}); + const content = fresh ? `${cwd}\n${sessionFile}\nfresh\n` : `${cwd}\n${sessionFile}\n`; + // Synchronous + best-effort. Infrequent (session create/switch/reset, never + // per-append), and writing in order matters: a lazy `/new` fresh crumb is + // re-stamped non-fresh the instant the session materializes, so an async + // fire-and-forget could land the two writes out of order and leave a + // materialized session marked fresh. + try { + fs.mkdirSync(breadcrumbDir, { recursive: true }); + fs.writeFileSync(breadcrumbFile, content); + } catch (err) { + if (!isEnoent(err)) logger.debug("Terminal breadcrumb write failed", { err }); + } } export interface TerminalBreadcrumb { cwd: string; sessionFile: string; + /** The recorded session file exists on disk right now. */ + exists: boolean; + /** Recorded as a `/new` fresh-session boundary whose JSONL may not exist yet. */ + fresh: boolean; } /** * Read the raw terminal breadcrumb for the current terminal. - * Returns the recorded cwd + session file (verified to exist) regardless of - * whether the recorded cwd still matches the current one. Callers decide how - * to interpret a cwd mismatch (e.g. a moved/renamed worktree). + * Returns the recorded cwd + session file regardless of whether the recorded + * cwd still matches the current one. Callers decide how to interpret a cwd + * mismatch (e.g. a moved/renamed worktree). + * + * A missing target file yields `null` UNLESS the breadcrumb is a `fresh` + * boundary — a lazy `/new` session whose JSONL was never written — in which case + * the entry is returned with `exists:false` so the caller can distinguish it + * from a genuinely stale/deleted breadcrumb. */ export async function readTerminalBreadcrumbEntry(): Promise { const terminalId = getTerminalId(); @@ -181,10 +207,13 @@ export async function readTerminalBreadcrumbEntry(): Promise= BLOB_EXTERNALIZE_THRESHOLD + ) { + return { ...obj, result: externalizeImageDataSync(blobStore, obj.result) }; + } if (shouldExternalizeImagePayload(obj, key)) { return { ...obj, data: externalizeImageDataSync(blobStore, obj.data, obj.mimeType) }; } diff --git a/packages/coding-agent/src/session/streaming-output.ts b/packages/coding-agent/src/session/streaming-output.ts index 449b57f6b..72d486029 100644 --- a/packages/coding-agent/src/session/streaming-output.ts +++ b/packages/coding-agent/src/session/streaming-output.ts @@ -22,6 +22,7 @@ export const ARTIFACT_DEFAULT_MAX_BYTES = 0; export const ARTIFACT_DEFAULT_HEAD_BYTES = 3 * 1024 * 1024; // 3 MiB const NL = "\n"; +const CR = "\r"; const ELLIPSIS = "…"; // ============================================================================= @@ -43,6 +44,8 @@ export interface OutputSummary { columnDroppedBytes?: number; /** Number of distinct lines that hit the per-line column cap. */ columnTruncatedLines?: number; + /** Configured per-line column cap in effect (chars), when > 0. */ + columnMax?: number; /** Artifact ID for internal URL access (artifact://) when truncated */ artifactId?: string; } @@ -735,6 +738,7 @@ export class OutputSink { #truncated = false; #lastChunkTime = 0; #pendingChunk = ""; + #pendingCarriageReturn = false; #pendingChunkTimer: Timer | undefined; // Per-line column cap streaming state (persists across `push` calls so a @@ -800,12 +804,45 @@ export class OutputSink { this.#artifactTailBudget = Math.max(0, this.#artifactMaxBytes - this.#artifactHeadBudget); } + /** + * Converts carriage-return progress updates into line boundaries while + * collapsing CRLF to one newline. A trailing CR is held until the next + * chunk so split CRLF sequences do not create blank lines. + */ + #normalizeCarriageReturns(text: string): string { + if (text.length === 0 || (!this.#pendingCarriageReturn && !text.includes(CR))) return text; + + let cursor = 0; + let normalized = ""; + if (this.#pendingCarriageReturn) { + this.#pendingCarriageReturn = false; + normalized = NL; + if (text.startsWith(NL)) cursor = 1; + } + + while (cursor < text.length) { + const carriageReturn = text.indexOf(CR, cursor); + if (carriageReturn === -1) { + normalized += text.substring(cursor); + break; + } + normalized += text.substring(cursor, carriageReturn); + if (carriageReturn === text.length - 1) { + this.#pendingCarriageReturn = true; + break; + } + normalized += NL; + cursor = text.startsWith(NL, carriageReturn + 1) ? carriageReturn + 2 : carriageReturn + 1; + } + return normalized; + } + /** * Push a chunk of output. The buffer management and onChunk callback run * synchronously. File sink writes are deferred and serialized internally. */ push(chunk: string): void { - chunk = sanitizeWithOptionalSixelPassthrough(chunk, sanitizeText); + chunk = sanitizeWithOptionalSixelPassthrough(chunk, text => sanitizeText(this.#normalizeCarriageReturns(text))); // Throttled onChunk: coalesce chunks arriving inside the throttle window. // A timer flushes quiet tails at the throttle boundary; dump() catches a @@ -836,7 +873,6 @@ export class OutputSink { const capped = this.#maxColumns > 0 ? this.#applyColumnCap(chunk) : chunk; const cappedBytes = capped === chunk ? rawBytes : Buffer.byteLength(capped, "utf-8"); const cappedThisChunk = cappedBytes < rawBytes; - if (cappedThisChunk) this.#truncated = true; // Mirror RAW chunk to the artifact file so the on-disk record is the full // uncapped stream. Mirror triggers on: in-memory overflow OR this chunk's @@ -1136,6 +1172,7 @@ export class OutputSink { this.#columnDroppedBytes = 0; this.#columnTruncatedLines = 0; this.#pendingChunk = ""; + this.#pendingCarriageReturn = false; } #clearPendingChunkTimer(): void { @@ -1207,6 +1244,10 @@ export class OutputSink { } async dump(notice?: string): Promise { + if (this.#pendingCarriageReturn) { + this.#pendingCarriageReturn = false; + this.push(NL); + } const noticeLine = notice ? `[${notice}]\n` : ""; // Flush any chunk still held back by the throttle so the live preview @@ -1276,6 +1317,7 @@ export class OutputSink { elidedLines, columnDroppedBytes: this.#columnDroppedBytes > 0 ? this.#columnDroppedBytes : undefined, columnTruncatedLines: this.#columnTruncatedLines > 0 ? this.#columnTruncatedLines : undefined, + columnMax: this.#columnTruncatedLines > 0 ? this.#maxColumns : undefined, artifactId: this.#file?.artifactId, }; } diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index feca888b5..b564b5471 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -316,6 +316,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ allowArgs: true, getTuiAutocompleteDescription: runtime => { if (!runtime.ctx.loopModeEnabled) return "Loop: off"; + if (runtime.ctx.loopModePaused) return "Loop: paused"; if (runtime.ctx.loopLimit) return `Loop: on (${describeLoopLimitRuntime(runtime.ctx.loopLimit)})`; if (runtime.ctx.loopPrompt) return "Loop: on (repeating prompt)"; return "Loop: on (waiting for next prompt)"; @@ -1389,6 +1390,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ }, { name: "new", + aliases: ["clear"], description: "Start a new session", handleTui: async (_command, runtime) => { runtime.ctx.editor.setText(""); @@ -1704,6 +1706,11 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ } catch { return usage(`Directory does not exist: ${resolvedPath}`, runtime); } + try { + await runtime.settings.flush(); + } catch (err) { + return usage(`Failed to save pending settings: ${errorMessage(err)}`, runtime); + } try { await runtime.sessionManager.moveTo(resolvedPath); } catch (err) { @@ -2254,6 +2261,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ // listClaudePluginRoots re-reads from disk on next access. const projectPath = await resolveActiveProjectRegistryPath(runtime.ctx.sessionManager.getCwd()); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); + await runtime.ctx.refreshSkillState(); await runtime.ctx.refreshSlashCommandState(); resetCapabilities(); runtime.ctx.showStatus("Plugins reloaded."); @@ -2319,6 +2327,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ }, { name: "quit", + aliases: ["q"], description: "Quit the application", handleTui: shutdownHandlerTui, }, @@ -2601,6 +2610,7 @@ export async function executeBuiltinSlashCommand( reloadPlugins: async () => { const projectPath = await resolveActiveProjectRegistryPath(ctx.sessionManager.getCwd()); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); + await ctx.refreshSkillState(); await ctx.refreshSlashCommandState(); resetCapabilities(); }, diff --git a/packages/coding-agent/src/slash-commands/helpers/active-oauth-account.ts b/packages/coding-agent/src/slash-commands/helpers/active-oauth-account.ts index 00e20c544..af98856d6 100644 --- a/packages/coding-agent/src/slash-commands/helpers/active-oauth-account.ts +++ b/packages/coding-agent/src/slash-commands/helpers/active-oauth-account.ts @@ -5,6 +5,22 @@ function normalizeIdentityValue(value: unknown): string | undefined { return typeof value === "string" && value.trim() ? value.trim().toLowerCase() : undefined; } +/** + * Session marker label for an active OAuth identity: the base identifier + * (email → accountId → projectId) suffixed with the organization when present + * and distinct. Same-email Anthropic multi-org accounts share the base, so the + * org suffix is the only field that tells the session's quota pool apart — + * mirrors the account-list rows (`formatUsageReportAccount`) and login success. + * Returns `undefined` when no identifier is recoverable. + */ +export function formatActiveAccountLabel(identity: OAuthAccountIdentity | undefined): string | undefined { + if (!identity) return undefined; + const base = identity.email || identity.accountId || identity.projectId; + if (!base) return undefined; + const org = identity.orgName || identity.orgId; + return org && org !== base ? `${base} (${org})` : base; +} + /** * True when a single usage-limit column belongs to the given OAuth identity. * diff --git a/packages/coding-agent/src/slash-commands/helpers/usage-report.ts b/packages/coding-agent/src/slash-commands/helpers/usage-report.ts index 34794e259..88f999b4c 100644 --- a/packages/coding-agent/src/slash-commands/helpers/usage-report.ts +++ b/packages/coding-agent/src/slash-commands/helpers/usage-report.ts @@ -124,7 +124,9 @@ function renderUsageReports( ); lines.push(` ${renderAsciiBar(limit.amount.usedFraction)}`); if (limit.window?.resetsAt && limit.window.resetsAt > nowMs) { - lines.push(` resets in ${formatDuration(limit.window.resetsAt - nowMs)}`); + lines.push( + ` ${limit.window.resetLabel ?? "resets"} in ${formatDuration(limit.window.resetsAt - nowMs)}`, + ); } if (limit.notes && limit.notes.length > 0) lines.push( diff --git a/packages/coding-agent/src/slash-commands/types.ts b/packages/coding-agent/src/slash-commands/types.ts index 3dfbe66f1..f5fcb22b8 100644 --- a/packages/coding-agent/src/slash-commands/types.ts +++ b/packages/coding-agent/src/slash-commands/types.ts @@ -64,10 +64,10 @@ export interface SlashCommandRuntime { /** Re-advertise the available command list (no-op outside ACP). */ refreshCommands: () => Promise | void; /** - * Reload plugin state (caches, slash command registry, project registries) - * and re-emit available commands. Used by `/reload-plugins`, `/move`, and - * `/marketplace`/`/plugins` mutations so the session sees a consistent view - * after plugin or project-scope changes. + * Reload plugin state (caches, skills, slash command registry, project + * registries) and re-emit available commands. Used by `/reload-plugins`, + * `/move`, and `/marketplace`/`/plugins` mutations so the session sees a + * consistent view after plugin or project-scope changes. */ reloadPlugins: () => Promise; notifyTitleChanged?: () => Promise | void; diff --git a/packages/coding-agent/src/stt/recorder.ts b/packages/coding-agent/src/stt/recorder.ts index 2dc03de6f..9797a4f93 100644 --- a/packages/coding-agent/src/stt/recorder.ts +++ b/packages/coding-agent/src/stt/recorder.ts @@ -11,6 +11,8 @@ export interface RecordingHandle { } const isWindows = process.platform === "win32"; +const linuxFFmpegFormats = new Map(); +const ffmpegCaptureFlags = ["-hide_banner", "-loglevel", "error", "-nostats"]; /** * Returns available recording tools in priority order. @@ -36,6 +38,36 @@ async function detectWindowsAudioDevice(bin: string): Promise { return audioDevices[0]; } +async function ffmpegInputArgs(bin: string): Promise { + if (isWindows) { + return ["-f", "dshow", "-i", `audio=${await detectWindowsAudioDevice(bin)}`]; + } + if (process.platform === "darwin") { + return ["-f", "avfoundation", "-i", ":default"]; + } + + let format = linuxFFmpegFormats.get(bin); + if (!format) { + const result = await $`${bin} -hide_banner -demuxers`.quiet().nothrow(); + if (result.exitCode !== 0) { + const stderr = result.stderr.toString().trim(); + throw new Error( + `Could not inspect ffmpeg input formats (code ${result.exitCode}): ${stderr || "(no output)"}`, + ); + } + const demuxers = result.stdout.toString(); + if (/^\s*D\s+(?:d\s+)?pulse(?:\s|$)/m.test(demuxers)) { + format = "pulse"; + } else if (/^\s*D\s+(?:d\s+)?alsa(?:\s|$)/m.test(demuxers)) { + format = "alsa"; + } else { + throw new Error("ffmpeg supports neither PulseAudio nor ALSA input on Linux"); + } + linuxFFmpegFormats.set(bin, format); + } + return ["-f", format, "-i", "default"]; +} + // ── Recording implementations ────────────────────────────────────── async function startSoxRecording(bin: string, outputPath: string): Promise { @@ -44,7 +76,7 @@ async function startSoxRecording(bin: string, outputPath: string): Promise { - let args: string[]; - if (isWindows) { - const device = await detectWindowsAudioDevice(bin); - args = [ - bin, - "-f", - "dshow", - "-i", - `audio=${device}`, - "-ar", - "16000", - "-ac", - "1", - "-sample_fmt", - "s16", - "-y", - outputPath, - ]; - } else if (process.platform === "darwin") { - args = [ - bin, - "-f", - "avfoundation", - "-i", - ":default", - "-ar", - "16000", - "-ac", - "1", - "-sample_fmt", - "s16", - "-y", - outputPath, - ]; - } else { - args = [bin, "-f", "pulse", "-i", "default", "-ar", "16000", "-ac", "1", "-sample_fmt", "s16", "-y", outputPath]; - } + const args = [ + bin, + ...ffmpegCaptureFlags, + ...(await ffmpegInputArgs(bin)), + "-ar", + "16000", + "-ac", + "1", + "-sample_fmt", + "s16", + "-y", + outputPath, + ]; const proc = Bun.spawn(args, { stdin: "pipe", stdout: "pipe", - stderr: "ignore", + stderr: "pipe", }); await verifyProcessAlive(proc, "ffmpeg"); @@ -119,7 +127,7 @@ async function startFFmpegRecording(bin: string, outputPath: string): Promise { const proc = Bun.spawn([bin, "-f", "S16_LE", "-r", "16000", "-c", "1", outputPath], { stdout: "pipe", - stderr: "ignore", + stderr: "pipe", }); await verifyProcessAlive(proc, "arecord"); return { @@ -260,20 +268,19 @@ async function startPowerShellRecording(outputPath: string): Promise; +type RecorderProcess = Subprocess<"ignore" | "pipe", "pipe", "pipe">; async function verifyProcessAlive(proc: RecorderProcess, tool: string): Promise { await Bun.sleep(300); const exited = await Promise.race([proc.exited.then(code => code), Bun.sleep(0).then(() => "running" as const)]); - - if (exited !== "running") { - let stderr = ""; - if (proc.stderr && typeof proc.stderr !== "number") { - stderr = await new Response(proc.stderr as ReadableStream).text(); - } - throw new Error(`${tool} exited immediately (code ${exited}): ${stderr.trim() || "(no output)"}`); + if (exited === "running") { + void proc.stderr.pipeTo(new WritableStream()).catch(() => {}); + return; } + + const stderr = await new Response(proc.stderr).text(); + throw new Error(`${tool} exited immediately (code ${exited}): ${stderr.trim() || "(no output)"}`); } // ── Public API ───────────────────────────────────────────────────── @@ -415,12 +422,18 @@ async function streamingRecorderArgs(recorder: ResolvedRecorder): Promise { const args = await streamingRecorderArgs(recorder); logger.debug("Starting streaming audio recording", { tool: recorder.tool, bin: recorder.bin }); - const proc = Bun.spawn(args, { stdin: "pipe", stdout: "pipe", stderr: "ignore" }); + const proc = Bun.spawn(args, { stdin: "pipe", stdout: "pipe", stderr: "pipe" }); // Read s16le bytes off stdout, carrying any trailing odd byte across chunk // boundaries so a sample is never split. Runs until the process closes stdout. diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 8b442719e..88ddf2b0e 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -24,7 +24,6 @@ import projectPromptTemplate from "./prompts/system/project-prompt.md" with { ty import systemPromptTemplate from "./prompts/system/system-prompt.md" with { type: "text" }; import { normalizeConcurrencyLimit } from "./task/parallel"; import { usesCodexTaskPrompt } from "./task/prompt-policy"; -import { shortenPath } from "./tools/render-utils"; import { type ActiveRepoContext, resolveActiveRepoContext } from "./utils/active-repo-context"; import { formatLocalCalendarDate } from "./utils/local-date"; import { normalizePromptPath } from "./utils/prompt-path"; @@ -470,7 +469,7 @@ export interface BuildSystemPromptOptions { /** Pre-loaded context files (skips discovery if provided). */ contextFiles?: Array<{ path: string; content: string; depth?: number }>; /** Skills provided directly to system prompt construction. */ - skills?: Skill[]; + skills?: readonly Skill[]; /** Pre-loaded rulebook rules (descriptions, excluding TTSR and always-apply). */ rules?: Array<{ name: string; description?: string; path: string; globs?: string[] }>; /** Intent field name injected into every tool schema. If set, explains the field in the prompt. */ @@ -637,7 +636,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): totalLines: 0, agentsMdFiles: [], }); - const skillsPromise: Promise = + const skillsPromise: Promise = providedSkills !== undefined ? Promise.resolve(providedSkills) : skillsSettings?.enabled !== false @@ -710,7 +709,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): const date = formatLocalCalendarDate(); const dateTime = date; - const promptCwd = shortenPath(normalizePromptPath(resolvedCwd)); + const promptCwd = normalizePromptPath(resolvedCwd); const activeRepoContextPrompt = renderActiveRepoContextPrompt(activeRepoContext); // Build tool metadata for system prompt rendering. @@ -729,13 +728,18 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): if (!toolPromptNames.has(mounted.name)) toolPromptNames.set(mounted.name, mounted.name); } const toolRefs = Object.fromEntries(toolPromptNames.entries()); - const toolInfo = toolNames.map(name => ({ + const xdevToolNames = new Set(xdevTools.map(mounted => mounted.name)); + // A direct custom tool can share a name with a retained built-in device. + // Presence in both toolNames and tools proves it still has a top-level definition. + const inventoryToolNames = + xdevToolNames.size === 0 ? toolNames : toolNames.filter(name => tools?.has(name) || !xdevToolNames.has(name)); + const toolInfo = inventoryToolNames.map(name => ({ name: toolPromptNames.get(name) ?? name, internalName: name, label: tools?.get(name)?.label ?? "", description: tools?.get(name)?.description ?? "", })); - const inventoryTools = toolNames.map(name => { + const inventoryTools = inventoryToolNames.map(name => { const meta = tools?.get(name); return { name: toolPromptNames.get(name) ?? name, diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 9a541dddf..6e0c53145 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -15,6 +15,7 @@ import { formatModelSelectorValue, formatModelStringWithRouting, resolveAgentPrewalkPattern, + resolveConfiguredModelPatterns, resolveModelOverride, resolveModelOverrideWithAuthFallback, } from "../config/model-resolver"; @@ -29,7 +30,6 @@ import { getSessionSlashCommands } from "../extensibility/extensions/get-command import { buildSkillPromptMessage, type Skill } from "../extensibility/skills"; import type { HindsightSessionState } from "../hindsight/state"; import type { LocalProtocolOptions } from "../internal-urls"; -import { callTool } from "../mcp/client"; import type { MCPManager } from "../mcp/manager"; import type { MnemopiSessionState } from "../mnemopi/state"; import subagentSystemPromptTemplate from "../prompts/system/subagent-system-prompt.md" with { type: "text" }; @@ -54,6 +54,7 @@ import type { EventBus } from "../utils/event-bus"; import { buildNamedToolChoice } from "../utils/tool-choice"; import type { WorkspaceTree } from "../workspace-tree"; import { generateTaskLabel } from "./label"; +import { resolveAgentPrewalkDefault } from "./prewalk"; import { subprocessToolRegistry } from "./subprocess-tool-registry"; import { type AgentDefinition, @@ -61,6 +62,9 @@ import { MAX_OUTPUT_BYTES, MAX_OUTPUT_LINES, type SingleResult, + type StructuredSubagentOutput, + type StructuredSubagentSchemaMode, + type StructuredSubagentSchemaSource, TASK_SUBAGENT_EVENT_CHANNEL, TASK_SUBAGENT_LIFECYCLE_CHANNEL, TASK_SUBAGENT_PROGRESS_CHANNEL, @@ -158,22 +162,44 @@ function resolveSubagentRetryFallbackCandidates( return candidates; } +function resolveSubagentDefaultRetryFallbackChain(settings: Settings): string[] | undefined { + const fallbackChain = settings.get("retry.fallbackChains")?.default; + if ( + !Array.isArray(fallbackChain) || + fallbackChain.length === 0 || + !fallbackChain.every(entry => typeof entry === "string") + ) { + return undefined; + } + return fallbackChain; +} + function installSubagentRetryFallbackChain(args: { settings: Settings; id: string; candidates: SubagentRetryFallbackCandidate[]; + defaultFallbackChain: string[] | undefined; model: Model | undefined; authFallbackUsed: boolean; }): string | undefined { - const { settings, id, candidates, model, authFallbackUsed } = args; - if (!model || authFallbackUsed || candidates.length <= 1) return undefined; + const { settings, id, candidates, defaultFallbackChain, model, authFallbackUsed } = args; + if (!model || authFallbackUsed || candidates.length === 0) return undefined; const selectedIndex = candidates.findIndex( candidate => candidate.model.provider === model.provider && candidate.model.id === model.id, ); if (selectedIndex < 0) return undefined; const fallbackSelectors = candidates.slice(selectedIndex + 1).map(candidate => candidate.selector); - if (fallbackSelectors.length === 0) return undefined; + const existingFallbackChains = settings.get("retry.fallbackChains"); + // A single explicit model may reuse a configured default chain, but never an implicit parent fallback. + const fallbackChain = fallbackSelectors.length > 0 ? fallbackSelectors : defaultFallbackChain; + if ( + !Array.isArray(fallbackChain) || + fallbackChain.length === 0 || + !fallbackChain.every(entry => typeof entry === "string") + ) { + return undefined; + } const role = `${SUBAGENT_RETRY_FALLBACK_ROLE_PREFIX}${id}`; const modelRoles: Record = {}; @@ -186,10 +212,10 @@ function installSubagentRetryFallbackChain(args: { } modelRoles[role] = candidates[selectedIndex].selector; settings.override("modelRoles", modelRoles); + // Insert the task-specific role first so another role assigned to the same model cannot capture fallback routing. const fallbackChains: Record = { - [role]: fallbackSelectors, + [role]: fallbackChain, }; - const existingFallbackChains = settings.get("retry.fallbackChains"); for (const existingRole in existingFallbackChains) { if (existingRole !== role) { fallbackChains[existingRole] = existingFallbackChains[existingRole]; @@ -292,7 +318,12 @@ export interface ExecutorOptions { */ parentActiveModelPattern?: string; thinkingLevel?: ConfiguredThinkingLevel; + /** Schema used to validate the final structured completion. */ outputSchema?: unknown; + /** Enforcement policy for {@link outputSchema}; defaults to legacy permissive behavior. */ + outputSchemaMode?: StructuredSubagentSchemaMode; + /** Origin of the selected schema, preserved in {@link SingleResult.structuredOutput}. */ + outputSchemaSource?: StructuredSubagentSchemaSource; /** * Caller supplied a schema that supersedes the agent's native output prompt. * Eval `agent(..., schema=...)` sets this so built-in agents ignore stale yield labels. @@ -307,7 +338,20 @@ export interface ExecutorOptions { * watchdog is already suspended for the call's duration. */ maxRuntimeMs?: number; + /** Include IRC only when the invocation policy permits collaboration. */ + enableIrc?: boolean; enableLsp?: boolean; + /** + * Enable MCP capabilities for this child. `false` suppresses both inherited + * MCP proxy tools and session MCP discovery; it never consults the + * process-global MCP manager. Defaults to `true`. + */ + enableMCP?: boolean; + /** + * Limit the child to its explicit host tool names and the required yield + * tool, suppressing discovered and always-included capabilities. + */ + restrictToolNames?: boolean; signal?: AbortSignal; onProgress?: (progress: AgentProgress) => void; /** @@ -450,6 +494,8 @@ interface FinalizeSubprocessOutputArgs { signalAborted: boolean; yieldItems?: YieldItem[]; outputSchema: unknown; + outputSchemaMode?: StructuredSubagentSchemaMode; + outputSchemaSource?: StructuredSubagentSchemaSource; lastAssistantText?: string; } @@ -459,6 +505,7 @@ interface FinalizeSubprocessOutputResult { stderr: string; abortedViaYield: boolean; hasYield: boolean; + structuredOutput?: StructuredSubagentOutput; } export const SUBAGENT_WARNING_SCHEMA_OVERRIDDEN = "SYSTEM WARNING: Subagent exhausted schema-retry budget; result was accepted despite failing the output schema."; @@ -494,6 +541,10 @@ function buildSchemaViolationOutcome( export function finalizeSubprocessOutput(args: FinalizeSubprocessOutputArgs): FinalizeSubprocessOutputResult { let { rawOutput, exitCode, stderr } = args; const { yieldItems, doneAborted, signalAborted, outputSchema, lastAssistantText } = args; + const mode = args.outputSchemaMode ?? "permissive"; + const source = args.outputSchemaSource ?? (outputSchema === undefined ? "none" : "session"); + const includeStructuredOutput = source !== "none"; + let structuredOutput: StructuredSubagentOutput | undefined; let abortedViaYield = false; const hasYield = Array.isArray(yieldItems) && yieldItems.length > 0; const hadFailureBeforeYield = exitCode !== 0 && stderr.trim().length > 0; @@ -514,15 +565,35 @@ export function finalizeSubprocessOutput(args: FinalizeSubprocessOutputArgs): Fi if (!assembled || assembled.missingData) { rawOutput = rawOutput ? `${SUBAGENT_WARNING_NULL_YIELD}\n\n${rawOutput}` : SUBAGENT_WARNING_NULL_YIELD; } else { - const { validator, error: schemaError } = buildOutputValidator(outputSchema); + const { validator, error: schemaError, normalized } = buildOutputValidator(outputSchema); const completeData = assembled.rawText ? assembled.data : parseStringifiedJson(assembled.data ?? null); - const result = - schemaError || assembled.schemaOverridden - ? { success: true as const } - : (validator?.validate(completeData) ?? { success: true as const }); - if (!result.success) { - const summary = summarizeValidationFailure(result, completeData, validator?.requiredFields ?? []); - const outcome = buildSchemaViolationOutcome(summary, completeData); + const validation = validator?.validate(completeData); + const failure = + validation && !validation.success + ? summarizeValidationFailure(validation, completeData, validator?.requiredFields ?? []) + : assembled.schemaOverridden + ? { message: SUBAGENT_WARNING_SCHEMA_OVERRIDDEN, missingRequired: [] } + : schemaError + ? { message: `invalid output schema: ${schemaError}`, missingRequired: [] } + : undefined; + if (includeStructuredOutput) { + structuredOutput = + schemaError || normalized === undefined + ? { + source, + mode, + status: "unavailable", + data: completeData, + error: schemaError ? `invalid output schema: ${schemaError}` : undefined, + } + : failure + ? { source, mode, status: "invalid", data: completeData, error: failure.message } + : { source, mode, status: "valid", data: completeData }; + } + const mustReject = + failure !== undefined && (mode === "strict" || (!assembled.schemaOverridden && !schemaError)); + if (mustReject && failure) { + const outcome = buildSchemaViolationOutcome(failure, completeData); rawOutput = outcome.rawOutput; stderr = outcome.stderr; exitCode = outcome.exitCode; @@ -540,9 +611,7 @@ export function finalizeSubprocessOutput(args: FinalizeSubprocessOutputArgs): Fi exitCode = 0; stderr = assembled.schemaOverridden ? SUBAGENT_WARNING_SCHEMA_OVERRIDDEN - : schemaError - ? `invalid output schema: ${schemaError}` - : ""; + : (structuredOutput?.error ?? ""); } else if (!stderr) { stderr = "Subagent failed after yielding a result."; } @@ -560,11 +629,22 @@ export function finalizeSubprocessOutput(args: FinalizeSubprocessOutputArgs): Fi const result = validator?.validate(completeData) ?? { success: true as const }; if (!result.success) { const summary = summarizeValidationFailure(result, completeData, validator?.requiredFields ?? []); + if (includeStructuredOutput) { + structuredOutput = { source, mode, status: "invalid", data: completeData, error: summary.message }; + } const outcome = buildSchemaViolationOutcome(summary, completeData); rawOutput = outcome.rawOutput; stderr = outcome.stderr; exitCode = outcome.exitCode; } else { + if (includeStructuredOutput) { + structuredOutput = { + source, + mode, + status: "valid", + data: completeData, + }; + } try { rawOutput = JSON.stringify(completeData, null, 2) ?? "null"; } catch (err) { @@ -587,7 +667,7 @@ export function finalizeSubprocessOutput(args: FinalizeSubprocessOutputArgs): Fi } } - return { rawOutput, exitCode, stderr, abortedViaYield, hasYield }; + return { rawOutput, exitCode, stderr, abortedViaYield, hasYield, structuredOutput }; } /** @@ -649,46 +729,54 @@ function getUsageTokens(usage: unknown): number { /** * Create proxy tools that reuse the parent's MCP connections. + * + * Each proxy delegates to the current source `MCPTool`/`DeferredMCPTool` rather + * than rebuilding a raw `tools/call` request, so the Task/subagent path shares + * the source tool's authoritative outbound boundary: harness-intent (`i`) + * stripping, optional-placeholder pruning, local-URL resolution, reconnect + * retry, abort handling, and result/provider metadata. The source tool is + * re-resolved on every call by raw MCP server/tool metadata (not the normalized + * display name), so a reconnect that swaps the instance in `getTools()` is + * always honored. The proxy adds only the Task-specific 60s call timeout, + * combining its abort signal with the caller's around source execution. */ export function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] { return mcpManager.getTools().map(tool => { - const mcpTool = tool as { mcpToolName?: string; mcpServerName?: string }; + const serverName = tool.mcpServerName ?? ""; + const mcpToolName = tool.mcpToolName ?? ""; return { name: tool.name, label: tool.label ?? tool.name, description: tool.description ?? "", parameters: tool.parameters, - execute: async (_toolCallId, params, _onUpdate, _ctx, signal) => { + strict: tool.strict, + mcpServerName: serverName, + mcpToolName, + execute: async (toolCallId, params, onUpdate, ctx, signal) => { if (signal?.aborted) { throw new ToolAbortError(); } - const serverName = mcpTool.mcpServerName ?? ""; - const mcpToolName = mcpTool.mcpToolName ?? ""; + // Re-resolve by raw MCP metadata so a reconnect that replaced the + // source instance is picked up; the display name alone is not enough. + const source = mcpManager + .getTools() + .find(t => t.mcpServerName === serverName && t.mcpToolName === mcpToolName); + if (!source?.execute) { + return { + content: [{ type: "text" as const, text: `MCP error: tool ${mcpToolName} no longer available` }], + details: { serverName, mcpToolName, isError: true }, + }; + } try { const timeoutController = new AbortController(); const timeoutSignal = timeoutController.signal; const combinedSignal = signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal; - const result = await withAbortTimeout( - (async () => { - const connection = await untilAborted(combinedSignal, () => - mcpManager.waitForConnection(serverName), - ); - return callTool(connection, mcpToolName, params as Record, { - signal: combinedSignal, - }); - })(), + return await withAbortTimeout( + Promise.resolve(source.execute(toolCallId, params, onUpdate, ctx, combinedSignal)), MCP_CALL_TIMEOUT_MS, signal, timeoutController, ); - return { - content: (result.content ?? []).map(item => - item.type === "text" - ? { type: "text" as const, text: item.text ?? "" } - : { type: "text" as const, text: JSON.stringify(item) }, - ), - details: { serverName, mcpToolName, isError: result.isError }, - }; } catch (error) { if (error instanceof ToolAbortError) { throw error; @@ -748,6 +836,8 @@ export function createSubagentSettings( export type AbortReason = "signal" | "terminate" | "timeout" | "budget"; +const MAX_YIELD_TOOL_ERRORS = 6; + /** Inputs for the run monitor driving one subagent assignment. */ interface RunMonitorArgs { index: number; @@ -794,11 +884,13 @@ interface SubagentRunMonitor { waitForBudgetStop(): Promise; /** The abort kind for this run, when an abort was requested. */ abortKind(): AbortReason | undefined; + terminalError(): string | undefined; /** True when the abort carries a precise external reason (signal / wall-clock / budget). */ hasExplicitAbortReason(): boolean; /** Whether the (attempted) abort counts as a cancelled run rather than an internal failure. */ isAbortedRun(): boolean; requestAbort(reason: AbortReason): void; + failWithError(message: string): void; abortActiveSession(): Promise; waitForActiveSessionAbort(): Promise; resolveSignalAbortReason(): string; @@ -857,7 +949,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { const finalOutputChunks: string[] = []; const RECENT_OUTPUT_TAIL_BYTES = 8 * 1024; let recentOutputTail = ""; - let tailLastLineRepresentable = false; + let recentOutputDirty = false; let resolved = false; let abortSent = false; let abortReason: AbortReason | undefined; @@ -885,6 +977,8 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { let budgetLimitExceeded = false; let budgetStopRequested = false; let budgetStopAbortPromise: Promise | undefined; + let terminalError: string | undefined; + let consecutiveYieldToolErrors = 0; let lastAssistantSalvageText: string | undefined; let activeSessionAbortPromise: Promise | undefined; @@ -941,6 +1035,11 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { : Promise.resolve(); }; + const failWithError = (message: string) => { + terminalError ??= message; + requestAbort("terminate"); + }; + // Handle abort signal if (signal) { signal.addEventListener( @@ -997,7 +1096,22 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { let lastProgressEmitMs = 0; let progressTimeoutId: NodeJS.Timeout | null = null; + // Recompute progress.recentOutput from the capped tail. Deferred: text_delta + // appends only extend the tail and mark it dirty; the (up to 8KB) split/filter + // runs synchronously here, immediately before the ONLY places the progress + // object is snapshotted ({...progress} for onProgress and the eventBus + // progress channel, both inside emitProgressNow — including the + // scheduleProgress(flush) finalize/error/cancel paths). Observers therefore + // always see exact state; no staleness beyond the existing 150ms coalescing. + const refreshRecentOutput = () => { + if (!recentOutputDirty) return; + recentOutputDirty = false; + const filtered = recentOutputTail.split("\n").filter(line => line.trim()); + progress.recentOutput = filtered.slice(-8).reverse(); + }; + const emitProgressNow = () => { + refreshRecentOutput(); progress.durationMs = Date.now() - startTime; onProgress?.({ ...progress }); const activityGist = @@ -1052,7 +1166,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { // failures just leave the label unset. const labelSource = assignment?.trim(); if (!args.description && args.modelRegistry && args.settings && labelSource) { - generateTaskLabel(labelSource, args.modelRegistry, args.settings, id) + generateTaskLabel(labelSource, args.modelRegistry, args.settings, id, abortSignal) .then(label => { if (!label || abortSignal.aborted || progress.description) return; progress.description = label; @@ -1080,36 +1194,16 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { return message.usage; }; - const updateRecentOutputLines = () => { - const lines = recentOutputTail.split("\n"); - const filtered = lines.filter(line => line.trim()); - progress.recentOutput = filtered.slice(-8).reverse(); - // The tail's last raw segment (after its final newline) is "represented" - // in recentOutput only when it trims non-empty — an empty/whitespace-only - // trailing segment is filtered out, so recentOutput[0] is then the line - // before it, not the tail's true last line. - tailLastLineRepresentable = lines[lines.length - 1].trim().length > 0; - }; - const appendRecentOutputTail = (text: string) => { if (!text) return; recentOutputTail += text; - const truncated = recentOutputTail.length > RECENT_OUTPUT_TAIL_BYTES; - if (truncated) { + if (recentOutputTail.length > RECENT_OUTPUT_TAIL_BYTES) { recentOutputTail = recentOutputTail.slice(-RECENT_OUTPUT_TAIL_BYTES); } - // Fast path: a token without a newline only extends the current last line. - // This runs on every text_delta token (hundreds/thousands per second while - // streaming), so skip re-splitting the whole (up to 8KB) tail unless the line - // structure actually changed. Requires no truncation AND the tail's last line - // already represented (trims non-empty) — otherwise boundaries shift and a - // full recompute is required. Appending to a non-empty line keeps it non-empty, - // so the flag stays valid across consecutive fast-path tokens. - if (truncated || text.includes("\n") || !tailLastLineRepresentable || progress.recentOutput.length === 0) { - updateRecentOutputLines(); - } else { - progress.recentOutput = [progress.recentOutput[0] + text, ...progress.recentOutput.slice(1)]; - } + // O(chunk) hot path: this runs on every text_delta token (hundreds/ + // thousands per second while streaming). Line reconstruction is deferred + // to refreshRecentOutput() at the emit boundary. + recentOutputDirty = true; }; const replaceRecentOutputFromContent = (content: unknown[]) => { @@ -1124,12 +1218,12 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { recentOutputTail = recentOutputTail.slice(-RECENT_OUTPUT_TAIL_BYTES); } } - updateRecentOutputLines(); + recentOutputDirty = true; }; const resetRecentOutput = () => { recentOutputTail = ""; - tailLastLineRepresentable = false; + recentOutputDirty = false; progress.recentOutput = []; }; @@ -1248,6 +1342,37 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { requestAbort("terminate"); } } + if (event.toolName === "yield") { + if (event.isError && !abortSent) { + consecutiveYieldToolErrors++; + let yieldErrorText = ""; + const resultContent = event.result?.content; + if (Array.isArray(resultContent)) { + const textParts: string[] = []; + for (const block of resultContent) { + if ( + block && + typeof block === "object" && + "type" in block && + block.type === "text" && + "text" in block && + typeof block.text === "string" + ) { + textParts.push(block.text); + } + } + yieldErrorText = textParts.join("\n").trim(); + } + if (consecutiveYieldToolErrors >= MAX_YIELD_TOOL_ERRORS) { + const suffix = yieldErrorText ? ` Last yield error: ${yieldErrorText}` : ""; + failWithError( + `Subagent submitted invalid yield results ${consecutiveYieldToolErrors} times; stopping to avoid an infinite submit loop.${suffix}`, + ); + } + } else if (!event.isError) { + consecutiveYieldToolErrors = 0; + } + } flushProgress = true; break; } @@ -1402,9 +1527,16 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { scheduleProgress(flushProgress); }; - const attach = (session: AgentSession): (() => void) => - session.subscribe(event => { + const attach = (session: AgentSession): (() => void) => { + let activeModel = session.model ? formatModelStringWithRouting(session.model) : undefined; + return session.subscribe(event => { emitSubagentEvent(event); + const nextModel = session.model ? formatModelStringWithRouting(session.model) : undefined; + if (nextModel && nextModel !== activeModel) { + activeModel = nextModel; + progress.resolvedModel = nextModel; + scheduleProgress(true); + } if (event.type === "auto_retry_start") { progress.retryState = { attempt: event.attempt, @@ -1446,15 +1578,18 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { } if (event.type === "retry_fallback_applied") { progress.resolvedModel = event.to; + progress.resolvedModelIsFallback = true; scheduleProgress(true); return; } if (event.type === "retry_fallback_succeeded") { progress.resolvedModel = event.model; + progress.resolvedModelIsFallback = true; scheduleProgress(true); return; } }); + }; const captureSalvage = (session: AgentSession): void => { // Best-effort salvage: capture the last assistant text so @@ -1483,6 +1618,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { hasUsage: () => hasUsage, yieldCalled: () => yieldCalled, runtimeLimitExceeded: () => runtimeLimitExceeded, + terminalError: () => terminalError, hasExplicitAbortReason: () => abortReason === "signal" || runtimeLimitExceeded || budgetLimitExceeded || budgetStopRequested, budgetStopRequested: () => budgetStopRequested, @@ -1493,6 +1629,7 @@ function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { isAbortedRun: () => abortReason === "signal" || runtimeLimitExceeded || budgetLimitExceeded || abortReason === undefined, requestAbort, + failWithError, abortActiveSession, waitForActiveSessionAbort, resolveSignalAbortReason, @@ -1688,6 +1825,7 @@ async function driveSessionToYield( } } } finally { + error ??= monitor.terminalError(); if (abortSignal.aborted && (!monitor.yieldCalled() || monitor.runtimeLimitExceeded())) { aborted = monitor.isAbortedRun(); if (aborted) { @@ -1710,6 +1848,8 @@ interface FinalizeRunArgs { assignment?: string; modelOverride?: string | string[]; outputSchema?: unknown; + outputSchemaMode?: StructuredSubagentSchemaMode; + outputSchemaSource?: StructuredSubagentSchemaSource; signal?: AbortSignal; artifactsDir?: string; eventBus?: EventBus; @@ -1747,6 +1887,8 @@ async function finalizeRunResult(args: FinalizeRunArgs): Promise { signalAborted: Boolean(signal?.aborted), yieldItems, outputSchema: args.outputSchema, + outputSchemaMode: args.outputSchemaMode, + outputSchemaSource: args.outputSchemaSource, lastAssistantText: monitor.lastAssistantSalvageText(), }); } finally { @@ -1841,6 +1983,7 @@ async function finalizeRunResult(args: FinalizeRunArgs): Promise { output: truncatedOutput, stderr, truncated: Boolean(truncated), + ...(finalized.structuredOutput ? { structuredOutput: finalized.structuredOutput } : {}), durationMs: Date.now() - args.startTime, tokens: progress.tokens, requests: progress.requests, @@ -1848,6 +1991,7 @@ async function finalizeRunResult(args: FinalizeRunArgs): Promise { contextWindow: progress.contextWindow, modelOverride, resolvedModel: progress.resolvedModel, + resolvedModelIsFallback: progress.resolvedModelIsFallback, error: exitCode !== 0 && stderr ? stderr : undefined, aborted: wasAborted, abortReason: finalAbortReason, @@ -1935,6 +2079,10 @@ export interface FollowUpTurnOptions { message: string; index?: number; description?: string; + /** Structured-output state retained from the original invocation. */ + outputSchema?: unknown; + outputSchemaMode?: StructuredSubagentSchemaMode; + outputSchemaSource?: StructuredSubagentSchemaSource; signal?: AbortSignal; onProgress?: (progress: AgentProgress) => void; eventBus?: EventBus; @@ -2019,6 +2167,9 @@ export async function runSubagentFollowUpTurn(options: FollowUpTurnOptions): Pro id, agent, task: message, + outputSchema: options.outputSchema, + outputSchemaMode: options.outputSchemaMode, + outputSchemaSource: options.outputSchemaSource, signal, artifactsDir: options.artifactsDir, eventBus: options.eventBus, @@ -2107,6 +2258,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise= 0 && childDepth >= maxRecursionDepth; + const ircEnabled = options.enableIrc !== false && isIrcEnabled(subagentSettings, childDepth); // Add tools if specified let toolNames: string[] | undefined; @@ -2121,9 +2273,9 @@ export async function runSubprocess(options: ExecutorOptions): Promise name !== "task"); } - // The hub is always available; the COOP prompt section advertises messaging, - // so a restricted whitelist must still carry `hub` for the subagent to use it. - if (toolNames && !toolNames.includes("hub")) { + // Ordinary agents retain the host's always-on collaboration capability. + // Restricted sessions must not widen their explicit host tool list with hub. + if (toolNames && !options.restrictToolNames && !toolNames.includes("hub")) { toolNames = [...toolNames, "hub"]; } if (toolNames?.includes("exec")) { @@ -2145,7 +2297,6 @@ export async function runSubprocess(options: ExecutorOptions): Promise { const subagentPrompt = prompt.render(subagentSystemPromptTemplate, { agent: agent.systemPrompt, @@ -2447,9 +2608,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise 0 ? mcpProxyTools : undefined, localProtocolOptions: options.localProtocolOptions, telemetry: subagentTelemetry, @@ -2524,6 +2686,8 @@ export async function runSubprocess(options: ExecutorOptions): Promise = new Set([ "rewind", ]); -const PLAN_MODE_AGENT_TOOL_ALLOWLIST: ReadonlySet = new Set(["ast_grep"]); - export function isReadOnlyAgent(agent: AgentDefinition): boolean { return !!agent.tools?.length && agent.tools.every(tool => READ_ONLY_TOOL_NAMES.has(tool)); } @@ -177,6 +161,7 @@ export function formatResultOutputFallback(result: Pick agent.blocking), @@ -218,14 +205,13 @@ function createTaskModeError(text: string): AgentToolResult { } /** - * Reject fields the current configuration does not accept. `schema` is never - * accepted (structured output comes from the agent definition's `output` - * frontmatter, the inherited session schema, or an eval-workflow - * `agent(..., schema)` call); `tasks`/`context` require `task.batch`. + * Reject legacy fields and shape/configuration combinations the current tool + * cannot accept. `outputSchema` is a first-class per-spawn field; stale + * `schema` remains an eval-only alias and is rejected. */ function validateShapeParams(batchEnabled: boolean, params: TaskParams): string | undefined { - if ((params as Record).schema !== undefined) { - return "The task tool does not accept `schema`. Rely on the selected agent definition's `output` schema or the inherited session schema; workflows needing ad-hoc structured output use eval `agent(prompt, schema)`."; + if (Object.hasOwn(params, "schema")) { + return "The task tool uses `outputSchema`; rename the stale `schema` field."; } if (!batchEnabled) { const disallowed = (["tasks", "context"] as const).filter(field => params[field] !== undefined); @@ -246,6 +232,19 @@ function validateShapeParams(batchEnabled: boolean, params: TaskParams): string * policy later, in `spawnParamsFor`. Returns a problem description, or * undefined when valid. */ +function hasInvalidModelSelector(model: unknown): boolean { + if (model === undefined) return false; + const selectors = typeof model === "string" ? [model] : Array.isArray(model) ? model : undefined; + const materializedSelectors = selectors ? Array.from(selectors) : []; + return ( + !selectors || + materializedSelectors.length === 0 || + materializedSelectors.some( + selector => typeof selector !== "string" || !selector.split(",").some(pattern => pattern.trim()), + ) + ); +} + function validateSpawnParams(params: TaskParams, batchEnabled: boolean): string | undefined { const hasTask = typeof params.task === "string" && params.task.trim() !== ""; const tasks = params.tasks; @@ -261,6 +260,9 @@ function validateSpawnParams(params: TaskParams, batchEnabled: boolean): string if (!item || typeof item.task !== "string" || item.task.trim() === "") { return `Task ${i + 1}${item?.name ? ` (\`${item.name}\`)` : ""} is missing \`task\`. Every task needs complete, self-contained instructions.`; } + if (hasInvalidModelSelector(item.model)) { + return `Task ${i + 1}${item.name ? ` (\`${item.name}\`)` : ""} has an invalid \`model\`. Provide a non-empty selector or a non-empty array of non-empty selectors.`; + } } const seen = new Map(); for (const item of tasks) { @@ -283,6 +285,9 @@ function validateSpawnParams(params: TaskParams, batchEnabled: boolean): string ? "Missing `tasks`. Provide a `tasks` array (one subagent per item) with a shared `context`." : "Missing `task`. Provide complete, self-contained instructions for the agent."; } + if (hasInvalidModelSelector(params.model)) { + return "Invalid `model`. Provide a non-empty selector or a non-empty array of non-empty selectors."; + } return undefined; } @@ -296,7 +301,9 @@ function resolveSpawnItems(params: TaskParams): TaskItem[] { if (Array.isArray(params.tasks) && params.tasks.length > 0) { return params.tasks; } - const item: TaskItem = { name: params.name, agent: params.agent, task: params.task }; + const item: TaskItem = { name: params.name, agent: params.agent, task: params.task, model: params.model }; + if ("outputSchema" in params) item.outputSchema = params.outputSchema; + if ("schemaMode" in params) item.schemaMode = params.schemaMode; if ("isolated" in params) item.isolated = params.isolated; return [item]; } @@ -314,7 +321,10 @@ function spawnParamsFor(params: TaskParams, item: TaskItem, defaultAgent: string const spawn: TaskParams = { agent: item.agent?.trim() || defaultAgent }; if (item.name !== undefined) spawn.name = item.name; if (item.task !== undefined) spawn.task = item.task; + if (item.model !== undefined) spawn.model = item.model; if (params.context !== undefined) spawn.context = params.context; + if ("outputSchema" in item) spawn.outputSchema = item.outputSchema; + if ("schemaMode" in item) spawn.schemaMode = item.schemaMode; if (item.isolated !== undefined) { spawn.isolated = item.isolated; } else if ("isolated" in params) { @@ -482,6 +492,14 @@ function discoverAgentsForCreate(cwd: string): Promise { return pending; } +function formatModelForApproval(model: unknown): string | undefined { + const selectors = typeof model === "string" ? [model] : Array.isArray(model) ? model : []; + const normalized = selectors.filter( + (selector): selector is string => typeof selector === "string" && !!selector.trim(), + ); + return normalized.length > 0 ? truncateForPrompt(normalized.join(" → ")) : undefined; +} + // ═══════════════════════════════════════════════════════════════════════════ // Tool Class // ═══════════════════════════════════════════════════════════════════════════ @@ -505,6 +523,8 @@ export class TaskTool implements AgentTool item.agent?.trim() || defaultAgent); + const normalizedSpawnParams = spawnItems.map(item => spawnParamsFor(params, item, defaultAgent)); + const resolvedAgents = normalizedSpawnParams.map(spawn => spawn.agent ?? defaultAgent); // Execution mode is per item: an item whose agent type declares // `blocking: true` runs inline on this turn (the parent waits on its // result); every other item becomes a background job when async // execution is available. - const itemBlocking = resolvedAgents.map( + const provisionalBlocking = resolvedAgents.map( name => this.#discoveredAgents.find(agent => agent.name === name)?.blocking === true, ); const asyncEnabled = this.session.settings.get("async.enabled"); const manager = asyncEnabled ? this.session.asyncJobManager : undefined; - const asyncItems = manager ? spawnItems.filter((_, index) => !itemBlocking[index]) : []; + const provisionalAsyncItems = manager ? spawnItems.filter((_, index) => !provisionalBlocking[index]) : []; const depthCapacity = canSpawnAtDepth( this.session.settings.get("task.maxRecursionDepth") ?? 2, this.session.taskDepth ?? 0, ); const ircEnabled = isIrcEnabled(this.session.settings, this.session.taskDepth ?? 0); + + if (!manager || provisionalAsyncItems.length === 0) { + // Sync fallback: async execution disabled, orphaned host that never + // wired a job manager, or every item's agent type declares + // `blocking: true`. `runStructuredSubagent` performs its own shared + // preflight before reserving an id in these inline paths. + if (asyncEnabled && !this.session.asyncJobManager) { + logger.warn("task: no AsyncJobManager registered; falling back to sync execution"); + } + const advisory = this.session.suppressSpawnAdvisory + ? undefined + : composeSpawnAdvisory({ + agents: resolvedAgents, + items: provisionalAsyncItems, + depthCapacity, + ircEnabled, + willRunAsync: false, + }); + const result = await this.#executeSyncFanout( + toolCallId, + params, + spawnItems.map((item, index) => ({ item, index })), + defaultAgent, + signal, + onUpdate, + ); + if (!advisory) return result; + let appended = false; + const content = result.content.map(part => { + if (!appended && part.type === "text" && typeof part.text === "string") { + appended = true; + return { ...part, text: `${part.text}\n\n${advisory}` }; + } + return part; + }); + if (!appended) content.push({ type: "text", text: advisory }); + return { ...result, content }; + } + + // Async jobs are otherwise registered before their body can reach + // `runStructuredSubagent`. Resolve the shared policy first so policy + // failures remain synchronous and cannot leave a queued invalid job. + const preflights = await Promise.all( + normalizedSpawnParams.map(async spawn => { + try { + return { policy: await this.#resolveSpawnPreflight(spawn) }; + } catch (error) { + return { error: error instanceof StructuredSubagentError ? error.message : String(error) }; + } + }), + ); + const preflightFailures = preflights + .map((preflight, index) => ("error" in preflight ? { index, error: preflight.error } : undefined)) + .filter((failure): failure is { index: number; error: string } => failure !== undefined); + const renderPreflightFailures = () => + preflightFailures + .map(({ index, error }) => { + const item = spawnItems[index]!; + return `Task ${item.name?.trim() || `#${index + 1}`} failed preflight: ${error}`; + }) + .join("\n"); + if (preflightFailures.length === spawnItems.length) { + return createTaskModeError(renderPreflightFailures()); + } + + const validIndices = preflights.flatMap((preflight, index) => (preflight.policy ? [index] : [])); + const validSpawns = validIndices.map(index => ({ item: spawnItems[index]!, index })); + const itemBlocking = preflights.map(preflight => preflight.policy?.effectiveAgent.blocking === true); + const asyncItems = validIndices.filter(index => !itemBlocking[index]).map(index => spawnItems[index]!); // Coordination only makes sense for spawns that keep running after this // call returns (the async subset). Blocking items have already completed // by then, so a "coordinate while they run" hint would misfire. - const willRunAsync = asyncItems.length > 0; const advisory = this.session.suppressSpawnAdvisory ? undefined : composeSpawnAdvisory({ - agents: resolvedAgents, + agents: validIndices.map(index => resolvedAgents[index]!), items: asyncItems, depthCapacity, ircEnabled, - willRunAsync, + willRunAsync: asyncItems.length > 0, }); // Returns a fresh result (copied content array, copied text part) rather // than mutating the caller's — task results are short-lived here, but an @@ -670,22 +799,35 @@ export class TaskTool implements AgentTool): AgentToolResult => { + if (preflightFailures.length === 0) return result; + const failures = renderPreflightFailures(); + let prepended = false; + const content = result.content.map(part => { + if (!prepended && part.type === "text" && typeof part.text === "string") { + prepended = true; + return { ...part, text: `${failures}\n\n${part.text}` }; + } + return part; + }); + if (!prepended) content.unshift({ type: "text", text: failures }); + return { ...result, content }; + }; + if (asyncItems.length === 0) { + return withPreflightFailures( + withAdvisory( + await this.#executeSyncFanout(toolCallId, params, validSpawns, defaultAgent, signal, onUpdate), + ), ); } - // Resolve agent ids up front so the immediate result can name them. - const outputManager = - this.session.agentOutputManager ?? new AgentOutputManager(this.session.getArtifactsDir ?? (() => null)); + // Async IDs are claimed before job registration, so retain the fallback + // manager on the session rather than recreating it for every call. + let outputManager = this.session.agentOutputManager; + if (!outputManager) { + outputManager = new AgentOutputManager(this.session.getArtifactsDir ?? (() => null)); + this.session.agentOutputManager = outputManager; + } const callStartedAt = Date.now(); const spawns: Array<{ agentId: string; @@ -694,10 +836,13 @@ export class TaskTool implements AgentTool = []; - for (let index = 0; index < spawnItems.length; index++) { - const item = spawnItems[index]; - const agentType = resolvedAgents[index]; - const agentSource = this.#discoveredAgents.find(agent => agent.name === agentType)?.source ?? "bundled"; + for (const index of validIndices) { + const item = spawnItems[index]!; + const agentType = resolvedAgents[index]!; + const preflight = preflights[index]!; + const policy = preflight.policy; + if (!policy) continue; + const agentSource = policy.agent.source; const agentId = await outputManager.allocate(item.name?.trim() || generateTaskName()); const assignment = (item.task ?? "").trim(); spawns.push({ @@ -733,7 +878,7 @@ export class TaskTool implements AgentTool 0 ? ` Failed to schedule ${failedSchedules.length} spawn${failedSchedules.length === 1 ? "" : "s"}: ${failedSchedules.join("; ")}.` : ""; - const coordinationHint = + const coordinationHint = [ started.length === 1 ? ircEnabled ? `DM \`${started[0].agentId}\` via \`hub\` send to coordinate while it runs; use \`hub\` only to inspect (\`jobs\`), wait, or cancel a stuck task.` : `Use \`hub\` to inspect (\`jobs\`), wait, or cancel a stuck task.` : ircEnabled ? `DM these ids via \`hub\` send to coordinate while they run; use \`hub\` only to inspect (\`jobs\`), wait, or cancel a stuck task.` - : `Use \`hub\` to inspect (\`jobs\`), wait, or cancel a stuck task by id.`; + : `Use \`hub\` to inspect (\`jobs\`), wait, or cancel a stuck task by id.`, + taskAsyncContractTemplate.trim(), + ].join("\n"); if (syncSpawns.length === 0) { if (spawns.length === 1) { @@ -814,30 +961,34 @@ export class TaskTool implements AgentTool `- \`${agentId}\` (job \`${jobId}\`)`).join("\n"); onUpdate?.({ content: [{ type: "text", text: `Spawned ${started.length} agents...` }], details: buildAsyncDetails(), }); - return withAdvisory({ - content: [ - { - type: "text", - text: `Spawned ${started.length} background agents using ${agentLabel}.${scheduleFailureSummary} Each result will be delivered when that agent yields.\n${startedListing}\n${coordinationHint}`, - }, - ], - details: buildAsyncDetails(), - }); + return withPreflightFailures( + withAdvisory({ + content: [ + { + type: "text", + text: `Spawned ${started.length} background agents using ${agentLabel}.${scheduleFailureSummary} Each result auto-delivers on yield unless a settled \`hub jobs\`/\`wait\` snapshot consumes it first.\n${startedListing}\n${coordinationHint}`, + }, + ], + details: buildAsyncDetails(), + }), + ); } // Mixed call: the async jobs above already run detached; the blocking @@ -861,7 +1012,7 @@ export class TaskTool implements AgentTool ({ item: spawn.item, index: spawn.index, preAllocatedId: spawn.agentId })), onItemProgress: onUpdate ? (index, progress) => { - const spawn = spawns[index]; + const spawn = spawns.find(candidate => candidate.index === index); if (spawn) spawn.progress = { ...progress, index }; onUpdate({ content: [{ type: "text", text: `Running ${syncLabel} inline...` }], @@ -897,15 +1048,17 @@ export class TaskTool implements AgentTool 0 - ? `Spawned ${started.length} background agent${started.length === 1 ? "" : "s"}.${scheduleFailureSummary} Each result will be delivered when that agent yields.\n${started.map(({ agentId, jobId }) => `- \`${agentId}\` (job \`${jobId}\`)`).join("\n")}\n${coordinationHint}` + ? `Spawned ${started.length} background agent${started.length === 1 ? "" : "s"}.${scheduleFailureSummary} Each result auto-delivers on yield unless a settled \`hub jobs\`/\`wait\` snapshot consumes it first.\n${started.map(({ agentId, jobId }) => `- \`${agentId}\` (job \`${jobId}\`)`).join("\n")}\n${coordinationHint}` : scheduleFailureSummary.trim(); const text = [merged.contentParts.join("\n\n"), spawnedSummary] .filter(section => section.trim().length > 0) .join("\n\n"); - return withAdvisory({ - content: [{ type: "text", text: text.length > 0 ? text : "No results." }], - details: buildAsyncDetails(), - }); + return withPreflightFailures( + withAdvisory({ + content: [{ type: "text", text: text.length > 0 ? text : "No results." }], + details: buildAsyncDetails(), + }), + ); } /** @@ -927,14 +1080,17 @@ export class TaskTool implements AgentTool { + const buildFollowUpHint = async (aborted: boolean): Promise => { if (aborted) { - const status = AgentRegistry.global().get(agentId)?.status; - if (status === "idle" || status === "parked") { + const ref = AgentRegistry.global().get(agentId); + const transcript = (await hasResolvableTranscript(agentId)) + ? `transcript at history://${agentId}` + : "transcript unavailable"; + if (ref?.status === "idle" || ref?.status === "parked") { const followUp = ircEnabled ? "message it via `hub` to resume; " : ""; - return `\n\n${agentId} was stopped but is still resumable — ${followUp}transcript at history://${agentId}`; + return `\n\n${agentId} was stopped but is still resumable — ${followUp}${transcript}`; } - return `\n\n${agentId} was aborted — transcript at history://${agentId}`; + return `\n\n${agentId} was aborted — ${transcript}`; } const followUp = ircEnabled ? "message it via `hub` to follow up; " : ""; return `\n\n${agentId} is now idle — ${followUp}transcript at history://${agentId}`; @@ -976,12 +1132,42 @@ export class TaskTool implements AgentTool, + ); + const forwardSyncProgress: AgentToolUpdateCallback = async update => { + const nextProgress = update.details?.progress?.[0]; + if (nextProgress) { + // The job body owns status and identity (id/index/agent); + // copy only the live metrics the subagent streams so the + // polling row reflects the resolved model, reasoning level, + // and running counters without reverting the "running" + // status back to the subagent's initial "pending" snapshot. + progress.resolvedModel = nextProgress.resolvedModel; + progress.resolvedModelIsFallback = nextProgress.resolvedModelIsFallback; + progress.tokens = nextProgress.tokens; + progress.requests = nextProgress.requests; + progress.contextTokens = nextProgress.contextTokens; + progress.contextWindow = nextProgress.contextWindow; + progress.cost = nextProgress.cost; + progress.toolCount = nextProgress.toolCount; + progress.currentTool = nextProgress.currentTool; + progress.lastIntent = nextProgress.lastIntent; + progress.recentTools = nextProgress.recentTools.slice(); + progress.recentOutput = nextProgress.recentOutput.slice(); + progress.retryState = nextProgress.retryState; + progress.retryFailure = nextProgress.retryFailure; + } + const updateText = + update.content.find(part => part.type === "text")?.text ?? `Running background task ${agentId}...`; + await reportProgress(updateText, buildDetails() as unknown as Record); + }; const result = await this.#executeSync( toolCallId, spawnParams, runSignal, - undefined, + forwardSyncProgress, agentId, progress.index, true, @@ -1002,12 +1188,19 @@ export class TaskTool implements AgentTool); + const deliveryText = `${finalText}${await buildFollowUpHint(singleResult?.aborted === true)}`; if (resultFailed) { // Mark the job itself failed; the failed agent stays interrogable. throw new TaskJobError(deliveryText); @@ -1021,9 +1214,9 @@ export class TaskTool implements AgentTool); const message = error instanceof Error ? error.message : String(error); - const hint = AgentRegistry.global().get(agentId) ? buildFollowUpHint(false) : ""; + const hint = AgentRegistry.global().get(agentId) ? await buildFollowUpHint(false) : ""; throw new TaskJobError(`${message}${hint}`); } finally { releasePermit(); @@ -1050,12 +1243,13 @@ export class TaskTool implements AgentTool, ): Promise> { - if (spawnItems.length === 1) { + if (spawns.length === 1) { + const spawn = spawns[0]!; const semaphore = this.#getSpawnSemaphore(); const invokedAt = Date.now(); await semaphore.acquire(signal); @@ -1063,11 +1257,11 @@ export class TaskTool implements AgentTool(); const emitCombined = () => { onUpdate?.({ - content: [{ type: "text", text: `Running ${spawnItems.length} agents...` }], + content: [{ type: "text", text: `Running ${spawns.length} agents...` }], details: { projectAgentsDir: null, results: [], @@ -1097,7 +1291,7 @@ export class TaskTool implements AgentTool ({ item, index })), + spawns, onItemProgress: onUpdate ? (index, progress) => { latestProgress.set(index, { ...progress, index }); @@ -1106,10 +1300,7 @@ export class TaskTool implements AgentTool ({ item, index })), - payloads, - ); + const merged = mergeSyncPayloads(spawns, payloads); return { content: [{ type: "text", text: merged.contentParts.join("\n\n") }], details: { @@ -1140,12 +1331,19 @@ export class TaskTool implements AgentTool | undefined)[]> { const { toolCallId, params, defaultAgent, spawns, signal, onItemProgress } = args; const semaphore = this.#getSpawnSemaphore(); - const { results } = await mapWithConcurrencyLimit( + const { results } = await mapWithConcurrencyLimitAllSettled( spawns, spawns.length, async (spawn, _position, workerSignal) => { const invokedAt = Date.now(); - await semaphore.acquire(workerSignal); + let semaphoreHeld = false; + try { + await semaphore.acquire(workerSignal); + semaphoreHeld = true; + } catch (error) { + if (workerSignal.aborted) return undefined; + throw error; + } const acquiredAt = Date.now(); try { const itemOnUpdate: AgentToolUpdateCallback | undefined = onItemProgress @@ -1165,12 +1363,26 @@ export class TaskTool implements AgentTool { + if (!settled) return undefined; + if (settled.status === "fulfilled") return settled.value; + const message = settled.reason instanceof Error ? settled.reason.message : String(settled.reason); + const item = spawns[position].item; + return { + content: [ + { + type: "text", + text: `Task ${item.name?.trim() || `#${spawns[position].index + 1}`} failed: ${message}`, + }, + ], + details: { projectAgentsDir: null, results: [], totalDurationMs: 0 }, + }; + }); } /** @@ -1204,351 +1416,60 @@ export class TaskTool implements AgentTool> { const startTime = Date.now(); - const { agents, projectAgentsDir } = await discoverAgents(this.session.cwd); - const agentName = params.agent ?? ""; - const sharedContext = this.#isBatchEnabled() ? params.context?.trim() || undefined : undefined; const assignment = (params.task ?? "").trim(); - const isolationMode = this.session.settings.get("task.isolation.mode"); - const isolationRequested = "isolated" in params ? params.isolated === true : false; - const isIsolated = isolationMode !== "none" && isolationRequested; - const mergeMode = this.session.settings.get("task.isolation.merge"); - const taskDepth = this.session.taskDepth ?? 0; - const subagentLspEnabled = (this.session.enableLsp ?? true) && this.session.settings.get("task.enableLsp"); - - if (isolationMode === "none" && "isolated" in params) { - return { - content: [{ type: "text", text: "Task isolation is disabled." }], - details: { projectAgentsDir, results: [], totalDurationMs: 0 }, - }; - } - - // Validate agent exists - const agent = getAgent(agents, agentName); - if (!agent) { - const available = agents.map(a => a.name).join(", ") || "none"; - return { - content: [{ type: "text", text: `Unknown agent "${agentName}". Available: ${available}` }], - details: { projectAgentsDir, results: [], totalDurationMs: 0 }, - }; - } - - // Check if agent is disabled in settings - const disabledAgents = this.session.settings.get("task.disabledAgents") as string[]; - if (disabledAgents.length > 0 && disabledAgents.includes(agentName)) { - const enabled = agents.filter(a => !disabledAgents.includes(a.name)).map(a => a.name); - return { - content: [ - { - type: "text", - text: `Agent "${agentName}" is disabled in settings. Enable it via /agents, or use a different agent type.${enabled.length > 0 ? ` Available: ${enabled.join(", ")}` : ""}`, - }, - ], - details: { projectAgentsDir, results: [], totalDurationMs: 0 }, - }; - } - - const planModeState = this.session.getPlanModeState?.(); - const planModeBaseTools = ["read", "grep", "glob", "lsp", "web_search"]; - const planModeTools = [ - ...planModeBaseTools, - ...(agent.tools ?? []).filter( - tool => PLAN_MODE_AGENT_TOOL_ALLOWLIST.has(tool) && !planModeBaseTools.includes(tool), - ), - ]; - const effectiveAgent: typeof agent = planModeState?.enabled - ? { - ...agent, - systemPrompt: `${planModeSubagentPrompt}\n\n${agent.systemPrompt}`, - tools: planModeTools, - spawns: undefined, - // Read-only exploration: never arm prewalk (its plan/implement - // nudges assume edit tools the plan-mode toolset doesn't have). - prewalk: undefined, - } - : agent; - - // Apply per-agent model override from settings (highest priority) - const agentModelOverrides = this.session.settings.get("task.agentModelOverrides"); - const settingsModelOverride = agentModelOverrides[agentName]; - const parentActiveModelPattern = this.session.getActiveModelString?.(); - const modelOverride = resolveAgentModelPatterns({ - settingsOverride: settingsModelOverride, - agentModel: effectiveAgent.model, - settings: this.session.settings, - activeModelPattern: parentActiveModelPattern, - fallbackModelPattern: this.session.getModelString?.(), - }); - const thinkingLevelOverride = effectiveAgent.thinkingLevel; - - // Output schema priority: agent frontmatter > inherited parent session. - // The task call itself never carries a schema; workflows needing ad-hoc - // structured output go through eval agent(prompt, schema). - const effectiveOutputSchema = effectiveAgent.output ?? this.session.outputSchema; - - let isolationContext: IsolationContext | null = null; - if (isIsolated) { - try { - isolationContext = await prepareIsolationContext(this.session.cwd); - } catch (err) { - const message = err instanceof Error ? err.message : String(err); - return { - content: [{ type: "text", text: `Isolated task execution requires a git repository. ${message}` }], - details: { projectAgentsDir, results: [], totalDurationMs: Date.now() - startTime }, - }; - } - } - const repoRoot = isolationContext?.repoRoot ?? null; - - const preferredIsolationBackend = parseIsolationMode(isolationMode); - - // Derive artifacts directory - const sessionFile = this.session.getSessionFile(); - const artifactsDir = sessionFile ? sessionFile.slice(0, -6) : null; - const tempArtifactsDir = artifactsDir ? null : path.join(os.tmpdir(), `omp-task-${Snowflake.next()}`); - const effectiveArtifactsDir = artifactsDir || tempArtifactsDir!; - - const localProtocolOptions: LocalProtocolOptions = this.session.localProtocolOptions ?? { - getArtifactsDir: this.session.getArtifactsDir ?? (() => null), - getSessionId: this.session.getSessionId ?? (() => null), - }; - - // Subagents adopt the parent's ArtifactManager so artifact IDs are unique - // across the whole tree and outputs land flat in the parent's dir. - const parentArtifactManager = this.session.getArtifactManager?.() ?? undefined; - - // When the session is executing an approved plan, hand the overall plan to - // every subagent so they share the main agent's plan context. Skipped in - // plan mode (read-only exploration uses planModeSubagentPrompt instead) and - // when no plan file exists at the session's reference path. - const planReference = planModeState?.enabled - ? undefined - : await loadOverallPlanReference( - this.session.getPlanReferencePath?.() ?? "local://PLAN.md", - localProtocolOptions, - ); - + const context = this.#isBatchEnabled() ? params.context?.trim() || undefined : undefined; + let latestProgress: AgentProgress | undefined; try { - // Check self-recursion prevention - if (this.#blockedAgent && agentName === this.#blockedAgent) { - return { - content: [ - { - type: "text", - text: `Cannot spawn ${this.#blockedAgent} agent from within itself (recursion prevention). Use a different agent type.`, - }, - ], - details: { projectAgentsDir, results: [], totalDurationMs: Date.now() - startTime }, - }; - } - - // Check spawn restrictions from parent - const spawnPolicy = resolveSpawnPolicy(this.session.getSessionSpawns()); - const spawnAllowed = - spawnPolicy.enabled && - (spawnPolicy.allowedAgents === null || spawnPolicy.allowedAgents.includes(agentName)); - if (!spawnAllowed) { - return { - content: [ - { type: "text", text: `Cannot spawn '${agentName}'. Allowed: ${spawnPolicy.allowedErrorText}` }, - ], - details: { projectAgentsDir, results: [], totalDurationMs: Date.now() - startTime }, - }; - } - - await fs.mkdir(effectiveArtifactsDir, { recursive: true }); - - // Allocate a unique ID across the session to prevent artifact collisions - let agentId: string; - if (preAllocatedId) { - agentId = preAllocatedId; - } else { - const outputManager = - this.session.agentOutputManager ?? new AgentOutputManager(this.session.getArtifactsDir ?? (() => null)); - agentId = await outputManager.allocate(params.name?.trim() || generateTaskName()); - } - - const availableSkills = [...(this.session.skills ?? [])]; - // Resolve autoload skills from agent definition against available skills - const resolvedAutoloadSkills = - agent.autoloadSkills?.length && availableSkills.length > 0 - ? agent.autoloadSkills - .map(name => availableSkills.find(s => s.name === name)) - .filter((s): s is NonNullable => s !== undefined) - : []; - const contextFiles = this.session.contextFiles?.filter( - file => path.basename(file.path).toLowerCase() !== "agents.md", - ); - const promptTemplates = this.session.promptTemplates; - const parentEvalSessionId = this.session.getEvalSessionId?.() ?? undefined; - const mcpManager = this.session.mcpManager ?? MCPManager.instance(); - - // Progress tracking for the single agent - let latestProgress: AgentProgress = { - index: spawnIndex, - id: agentId, - agent: agentName, - agentSource: agent.source, - status: "pending", - task: renderSubagentUserPrompt(assignment), + const execution = await runStructuredSubagent({ + session: this.session, + invocationKind: "task", assignment, - recentTools: [], - recentOutput: [], - toolCount: 0, - requests: 0, - tokens: 0, - cost: 0, - durationMs: 0, - modelOverride, - }; - const emitProgress = () => { - onUpdate?.({ - content: [{ type: "text", text: `Running agent ${agentId}...` }], - details: { - projectAgentsDir, - results: [], - totalDurationMs: Date.now() - startTime, - progress: [latestProgress], - }, - }); - }; - emitProgress(); - - const buildCommitMessageFn = makeIsolationCommitMessage(this.session); - - const sharedRunOptions = { - cwd: this.session.cwd, - agent: effectiveAgent, - task: renderSubagentUserPrompt(assignment), - assignment, - context: sharedContext, - planReference, + context, + agent: params.agent, + model: params.model, + ...(Object.hasOwn(params, "outputSchema") ? { outputSchema: params.outputSchema } : {}), + ...(Object.hasOwn(params, "schemaMode") ? { schemaMode: params.schemaMode } : {}), + identity: { id: preAllocatedId, label: params.name }, index: spawnIndex, parentToolCallId: toolCallId, detached, - id: agentId, - taskDepth, invokedAt: launchTiming?.invokedAt, acquiredAt: launchTiming?.acquiredAt, - modelOverride, - parentActiveModelPattern, - thinkingLevel: thinkingLevelOverride, - outputSchema: effectiveOutputSchema, - sessionFile, - persistArtifacts: !!artifactsDir, - artifactsDir: effectiveArtifactsDir, - enableLsp: subagentLspEnabled, + ...("isolated" in params ? { isolation: { requested: params.isolated } } : {}), + blockedAgent: this.#blockedAgent, + enableLsp: (this.session.enableLsp ?? true) && this.session.settings.get("task.enableLsp"), + enableIrc: isIrcEnabled(this.session.settings, this.session.taskDepth ?? 0), + maxRuntimeMs: this.session.settings.get("task.maxRuntimeMs"), signal, - eventBus: this.session.eventBus, - onProgress: (progress: AgentProgress) => { - // Shallow snapshot; recentTools is mutated in place by the - // executor, the rest is reassigned or immutable. A deep clone - // here cost O(extractedToolData) per progress event. + onProgress: progress => { latestProgress = { ...progress, recentTools: progress.recentTools.slice() }; - emitProgress(); + onUpdate?.({ + content: [{ type: "text", text: `Running agent ${progress.id}...` }], + details: { + projectAgentsDir: null, + results: [], + totalDurationMs: Date.now() - startTime, + progress: [latestProgress], + }, + }); }, - authStorage: this.session.authStorage, - modelRegistry: this.session.modelRegistry, - settings: this.session.settings, - mcpManager, - contextFiles, - skills: availableSkills, - autoloadSkills: resolvedAutoloadSkills, - workspaceTree: this.session.workspaceTree, - promptTemplates, - rules: this.session.rules, - preloadedExtensionPaths: this.session.extensionPaths, - preloadedCustomToolPaths: this.session.customToolPaths, - localProtocolOptions, - parentArtifactManager, - parentHindsightSessionState: this.session.getHindsightSessionState?.(), - parentMnemopiSessionState: this.session.getMnemopiSessionState?.(), - parentTelemetry: this.session.getTelemetry?.(), - parentEvalSessionId, - parentAgentId: this.session.getAgentId?.() ?? MAIN_AGENT_ID, - // Live source of truth for `tier.subagent: inherit`. When the session - // exposes a tier accessor, pass the per-family map or null (null = - // explicit none, e.g. /fast off); otherwise leave undefined so inherit - // falls back to the subagent's configured tier.* settings. - parentServiceTier: this.session.getServiceTierByFamily - ? (this.session.getServiceTierByFamily() ?? null) - : undefined, - }; - - const runTask = async (): Promise => { - if (!isIsolated) { - return runSubprocess(sharedRunOptions); - } - if (!isolationContext) { - throw new Error("Isolated task execution not initialized."); - } - const taskStart = Date.now(); - return runIsolatedSubprocess({ - baseOptions: sharedRunOptions, - context: isolationContext, - preferredBackend: preferredIsolationBackend, - agentId, - mergeMode, - artifactsDir: effectiveArtifactsDir, - buildCommitMessage: buildCommitMessageFn, - buildFailureResult: err => { - const message = err instanceof Error ? err.message : String(err); - return { - index: spawnIndex, - id: agentId, - agent: agent.name, - agentSource: agent.source, - task: renderSubagentUserPrompt(assignment), - assignment, - exitCode: 1, - output: "", - stderr: message, - truncated: false, - durationMs: Date.now() - taskStart, - tokens: 0, - requests: 0, - modelOverride, - error: message, - }; - }, - }); - }; - - const result = await runTask(); - - let mergeSummary = ""; - let changesApplied: boolean | null = null; - let mergedBranchForNestedPatches = false; - if (isIsolated && repoRoot) { - const outcome = await mergeIsolatedChanges({ result, repoRoot, mergeMode }); - mergeSummary = outcome.summary; - changesApplied = outcome.changesApplied; - mergedBranchForNestedPatches = outcome.mergedBranchForNestedPatches; - } - - // Apply nested repo patches (separate from parent git). - if (isIsolated && repoRoot) { - mergeSummary += await applyEligibleNestedPatches({ - result, - repoRoot, - mergeMode, - changesApplied, - mergedBranchForNestedPatches, - commitMessage: buildCommitMessageFn(), - }); - } - - // Cleanup temp directory if used - const shouldCleanupTempArtifacts = - tempArtifactsDir && (!isIsolated || changesApplied === true || changesApplied === null); - if (shouldCleanupTempArtifacts) { - await fs.rm(tempArtifactsDir, { recursive: true, force: true }); - } - - return this.#buildResultPayload(result, projectAgentsDir, Date.now() - startTime, mergeSummary); - } catch (err) { + }); + return this.#buildResultPayload( + execution.result, + execution.policy.discovery.projectAgentsDir, + Date.now() - startTime, + execution.mergeSummary, + ); + } catch (error) { + const message = error instanceof StructuredSubagentError ? error.message : String(error); return { - content: [{ type: "text", text: `Task execution failed: ${err}` }], - details: { projectAgentsDir, results: [], totalDurationMs: Date.now() - startTime }, + content: [{ type: "text", text: `Task execution failed: ${message}` }], + details: { + projectAgentsDir: null, + results: [], + totalDurationMs: Date.now() - startTime, + ...(latestProgress ? { progress: [latestProgress] } : {}), + }, }; } } diff --git a/packages/coding-agent/src/task/label.ts b/packages/coding-agent/src/task/label.ts index 665dfe0a6..6c886b9e6 100644 --- a/packages/coding-agent/src/task/label.ts +++ b/packages/coding-agent/src/task/label.ts @@ -15,6 +15,7 @@ export async function generateTaskLabel( registry: ModelRegistry, settings: Settings, sessionId?: string, + signal?: AbortSignal, ): Promise { const text = assignment.trim(); if (!text) return null; @@ -27,6 +28,7 @@ export async function generateTaskLabel( undefined, undefined, TASK_LABEL_SYSTEM_PROMPT, + signal, ); } catch (err) { logger.debug("task-label: generation failed", { diff --git a/packages/coding-agent/src/task/parallel.ts b/packages/coding-agent/src/task/parallel.ts index 219822e50..98029fa42 100644 --- a/packages/coding-agent/src/task/parallel.ts +++ b/packages/coding-agent/src/task/parallel.ts @@ -83,6 +83,49 @@ export async function mapWithConcurrencyLimit( return { results, aborted: signal?.aborted ?? false }; } +/** Result of a concurrency-limited operation that waits for every launched item. */ +export interface ParallelSettledResult { + /** Settled results in original input order; absent entries were never launched after cancellation. */ + results: (PromiseSettledResult | undefined)[]; + /** Whether cancellation prevented scheduling all items. */ + aborted: boolean; +} + +/** + * Execute items with a concurrency limit without failing fast. Rejections are + * captured at their input position and already launched siblings always settle + * before this function returns. Cancellation stops new launches but preserves + * the settled state of every item that began. + */ +export async function mapWithConcurrencyLimitAllSettled( + items: T[], + concurrency: number, + fn: (item: T, index: number, signal: AbortSignal) => Promise, + signal?: AbortSignal, +): Promise> { + const normalizedConcurrency = Number.isFinite(concurrency) ? Math.floor(concurrency) : items.length; + const effectiveConcurrency = normalizedConcurrency > 0 ? normalizedConcurrency : items.length; + const limit = Math.max(1, Math.min(effectiveConcurrency, items.length)); + const results: (PromiseSettledResult | undefined)[] = new Array(items.length); + const workerSignal = signal ?? new AbortController().signal; + let nextIndex = 0; + + const worker = async (): Promise => { + while (!workerSignal.aborted) { + const index = nextIndex++; + if (index >= items.length) return; + try { + results[index] = { status: "fulfilled", value: await fn(items[index], index, workerSignal) }; + } catch (reason) { + results[index] = { status: "rejected", reason }; + } + } + }; + + await Promise.all(Array.from({ length: limit }, () => worker())); + return { results, aborted: workerSignal.aborted }; +} + /** * Simple counting semaphore for limiting concurrency across independently-scheduled async work. * diff --git a/packages/coding-agent/src/task/persisted-revive.ts b/packages/coding-agent/src/task/persisted-revive.ts index 9813b707e..bd073c225 100644 --- a/packages/coding-agent/src/task/persisted-revive.ts +++ b/packages/coding-agent/src/task/persisted-revive.ts @@ -79,9 +79,10 @@ export function createPersistedSubagentReviverFactory( }); const artifactManager = ctx.session.sessionManager.getArtifactManager(); if (artifactManager) reopened.adoptArtifactManager(artifactManager); - // Reuse the parent's live MCP connections via proxy tools (no - // re-discovery), exactly as the executor does for live subagents. - const mcpManager = MCPManager.instance(); + // A restricted persisted contract must not consult process-global MCP + // state: same-name MCP tools are untrusted capability sources. + const restrictToolNames = init.restrictToolNames === true; + const mcpManager = restrictToolNames ? undefined : MCPManager.instance(); const mcpProxyTools = mcpManager ? createMCPProxyTools(mcpManager) : []; const { session } = await createAgentSession({ cwd: ctx.session.sessionManager.getCwd(), @@ -99,16 +100,27 @@ export function createPersistedSubagentReviverFactory( taskDepth, toolNames: init.tools, outputSchema: init.outputSchema, + outputSchemaMode: init.outputSchemaMode, + restrictToolNames: restrictToolNames || undefined, requireYieldTool: true, systemPrompt: () => [init.systemPrompt], // Old files predate persisted spawns: deny re-spawning rather than let // createAgentSession default to wildcard ("*"). spawns: init.spawns ?? "", hasUI: false, - enableLsp: ctx.enableLsp, - enableMCP: !mcpManager, - mcpManager, - customTools: mcpProxyTools.length > 0 ? mcpProxyTools : undefined, + enableLsp: restrictToolNames ? false : ctx.enableLsp, + ...(restrictToolNames + ? { + enableIrc: false, + enableMCP: false, + preloadedExtensionPaths: [], + preloadedCustomToolPaths: [], + } + : { + enableMCP: !mcpManager, + mcpManager, + customTools: mcpProxyTools.length > 0 ? mcpProxyTools : undefined, + }), }); // Clamp the active set to the persisted list: createAgentSession's // `alwaysInclude` can re-add non-defaultInactive extension/custom tools diff --git a/packages/coding-agent/src/task/prewalk.ts b/packages/coding-agent/src/task/prewalk.ts new file mode 100644 index 000000000..1444b62bb --- /dev/null +++ b/packages/coding-agent/src/task/prewalk.ts @@ -0,0 +1,6 @@ +import type { AgentDefinition } from "./types"; + +/** Resolve an agent's prewalk default, including the bundled task opt-in. */ +export function resolveAgentPrewalkDefault(agent: AgentDefinition, taskPrewalk: boolean): boolean | string | undefined { + return agent.prewalk ?? (taskPrewalk && agent.source === "bundled" && agent.name === "task" ? true : undefined); +} diff --git a/packages/coding-agent/src/task/structured-subagent.ts b/packages/coding-agent/src/task/structured-subagent.ts new file mode 100644 index 000000000..fe9e8d026 --- /dev/null +++ b/packages/coding-agent/src/task/structured-subagent.ts @@ -0,0 +1,644 @@ +/** + * Shared policy resolution and execution for task and eval subagents. + * + * The two public frontends deliberately retain their presentation concerns, but + * every decision that affects what a child may run lives here. + */ +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import path from "node:path"; +import { $env, prompt, Snowflake } from "@oh-my-pi/pi-utils"; +import { resolveAgentModelPatterns } from "../config/model-resolver"; +import type { LocalProtocolOptions } from "../internal-urls"; +import { registerArtifactsDir } from "../internal-urls/registry-helpers"; +import { MCPManager } from "../mcp/manager"; +import { loadOverallPlanReference } from "../plan-mode/plan-handoff"; +import planModeSubagentPrompt from "../prompts/system/plan-mode-subagent.md" with { type: "text" }; +import subagentUserPromptTemplate from "../prompts/system/subagent-user-prompt.md" with { type: "text" }; +import { MAIN_AGENT_ID } from "../registry/agent-registry"; +import type { ToolSession } from "../tools"; +import { isIrcEnabled } from "../tools/hub"; +import { buildOutputValidator } from "../tools/output-schema-validator"; +import { type DiscoveryResult, discoverAgents, getAgent } from "./discovery"; +import { type ExecutorOptions, runSubprocess } from "./executor"; +import { + applyEligibleNestedPatches, + type IsolationContext, + makeIsolationCommitMessage, + mergeIsolatedChanges, + prepareIsolationContext, + runIsolatedSubprocess, +} from "./isolation-runner"; +import { generateTaskName } from "./name-generator"; +import { AgentOutputManager } from "./output-manager"; +import { resolveSpawnPolicy } from "./spawn-policy"; +import { + type AgentDefinition, + type AgentProgress, + canSpawnAtDepth, + type SingleResult, + type StructuredSubagentOutput, +} from "./types"; +import { type NestedRepoPatch, parseIsolationMode } from "./worktree"; + +/** Validation behavior requested for an effective output schema. */ +export type StructuredSubagentSchemaMode = "permissive" | "strict"; + +/** Where an effective output schema came from. */ +export type StructuredSubagentSchemaSource = "caller" | "agent" | "session" | "none"; + +/** Final structured completion metadata returned for a schema-bearing run. */ +export type StructuredSubagentSchemaResult = StructuredSubagentOutput; + +/** A schema validation or extraction error attached to structured completion metadata. */ +export type StructuredSubagentSchemaError = NonNullable; + +/** A selected schema paired with its source and enforcement mode. */ +export interface StructuredSubagentSchemaResolution { + schema: unknown; + source: StructuredSubagentSchemaSource; + mode: StructuredSubagentSchemaMode; + outputSchemaOverridesAgent: boolean; +} + +/** Isolation controls shared by the task and eval surfaces. */ +export interface StructuredSubagentIsolationControls { + requested?: boolean; + merge?: "patch" | "branch"; + apply?: boolean; +} + +/** Identity and presentation metadata supplied by the calling surface. */ +export interface StructuredSubagentIdentity { + /** A previously reserved output/registry id. */ + id?: string; + /** Stable user-facing label used when allocating a new id. */ + label?: string; +} + +/** One normalized child invocation. */ +export interface StructuredSubagentRequest { + session: ToolSession; + invocationKind: "task" | "eval"; + assignment: string; + context?: string; + agent?: string; + model?: string | string[]; + /** Presence, rather than truthiness, makes this the highest-priority schema. */ + outputSchema?: unknown; + schemaMode?: StructuredSubagentSchemaMode; + identity?: StructuredSubagentIdentity; + index?: number; + parentToolCallId?: string; + detached?: boolean; + invokedAt?: number; + acquiredAt?: number; + isolation?: StructuredSubagentIsolationControls; + /** The parent agent name forbidden from recursively spawning itself. */ + blockedAgent?: string; + /** Preserve a completed temporary artifacts directory for an agent:// handle. */ + retainArtifacts?: boolean; + /** Task UI agents keep live registry references; eval one-shots normally do not. */ + keepAlive?: boolean; + /** Task subagents share their parent's eval kernel; eval bridge children must not. */ + shareEvalSession?: boolean; + /** Task frontends may inherit LSP; eval frontends normally set this false. */ + enableLsp?: boolean; + /** Explicitly pass false for plan mode or invocation kinds that must not use IRC. */ + enableIrc?: boolean; + /** `0` disables executor wall-clock timeout. Undefined inherits settings. */ + maxRuntimeMs?: number; + signal?: AbortSignal; + onProgress?: (progress: AgentProgress) => void; +} + +/** A normalized preflight result, reusable by tests and adapters. */ +export interface EffectiveSubagentPolicy { + discovery: DiscoveryResult; + agentName: string; + agent: AgentDefinition; + effectiveAgent: AgentDefinition; + modelOverride?: string | string[]; + parentActiveModelPattern?: string; + schema: StructuredSubagentSchemaResolution; + planMode: boolean; + isIsolated: boolean; + mergeMode: "patch" | "branch"; + applyChanges: boolean; + enableLsp: boolean; + enableIrc: boolean; +} + +/** Settled child execution plus data needed by the frontends' own rendering. */ +export interface StructuredSubagentResult { + result: SingleResult; + policy: EffectiveSubagentPolicy; + mergeSummary: string; + changesApplied: boolean | null; + artifactsDir: string; + temporaryArtifacts: boolean; +} + +/** Machine-readable failure category so adapters can retain their native errors. */ +export class StructuredSubagentError extends Error { + readonly kind: "preflight" | "isolation" | "execution"; + + constructor(kind: "preflight" | "isolation" | "execution", message: string, options?: ErrorOptions) { + super(message, options); + this.name = "StructuredSubagentError"; + this.kind = kind; + } +} + +const PLAN_MODE_TOOLS = ["read", "grep", "glob", "web_search"] as const; + +function renderSubagentPrompt(assignment: string): string { + return prompt.render(subagentUserPromptTemplate, { assignment: assignment.trim() }); +} + +function trimToUndefined(value: string | undefined): string | undefined { + const trimmed = value?.trim(); + return trimmed || undefined; +} + +function sanitizeAgentId(value: string | undefined): string | undefined { + const trimmed = trimToUndefined(value); + const sanitized = trimmed?.replace(/[^A-Za-z0-9_-]+/g, "").slice(0, 48); + return sanitized || undefined; +} + +function resolveSchema(request: StructuredSubagentRequest, agent: AgentDefinition): StructuredSubagentSchemaResolution { + const mode = request.schemaMode ?? request.session.outputSchemaMode ?? "permissive"; + if (Object.hasOwn(request, "outputSchema")) { + return { schema: request.outputSchema, source: "caller", mode, outputSchemaOverridesAgent: true }; + } + if (agent.output !== undefined) { + return { schema: agent.output, source: "agent", mode, outputSchemaOverridesAgent: false }; + } + if (request.session.outputSchema !== undefined) { + return { schema: request.session.outputSchema, source: "session", mode, outputSchemaOverridesAgent: false }; + } + return { schema: undefined, source: "none", mode, outputSchemaOverridesAgent: false }; +} + +function createPlanModeAgent(agent: AgentDefinition): AgentDefinition { + const tools = [...PLAN_MODE_TOOLS, ...(agent.tools ?? []).filter(tool => tool === "ast_grep")]; + return { + ...agent, + systemPrompt: `${planModeSubagentPrompt}\n\n${agent.systemPrompt}`, + tools, + spawns: undefined, + prewalk: undefined, + }; +} + +function assertPlanControlsAllowed(request: StructuredSubagentRequest, planMode: boolean): void { + if (!planMode) return; + const isolation = request.isolation; + if ( + isolation && + (Object.hasOwn(isolation, "requested") || Object.hasOwn(isolation, "apply") || Object.hasOwn(isolation, "merge")) + ) { + throw new StructuredSubagentError( + "preflight", + "Subagent isolation, apply, and merge controls are unavailable in plan mode.", + ); + } +} + +function assertDepthAndSpawnAllowed(request: StructuredSubagentRequest, agentName: string): void { + const taskDepth = request.session.taskDepth ?? 0; + const maxDepth = request.session.settings.get("task.maxRecursionDepth") ?? 2; + if (!canSpawnAtDepth(maxDepth, taskDepth)) { + throw new StructuredSubagentError( + "preflight", + `Cannot spawn another agent at task depth ${taskDepth}; maximum depth is ${maxDepth}.`, + ); + } + const blockedAgent = request.blockedAgent ?? $env.PI_BLOCKED_AGENT; + if (blockedAgent && blockedAgent === agentName) { + throw new StructuredSubagentError( + "preflight", + `Cannot spawn ${blockedAgent} agent from within itself (recursion prevention). Use a different agent type.`, + ); + } + const spawnPolicy = resolveSpawnPolicy(request.session.getSessionSpawns()); + if (!spawnPolicy.enabled || (spawnPolicy.allowedAgents !== null && !spawnPolicy.allowedAgents.includes(agentName))) { + throw new StructuredSubagentError( + "preflight", + `Cannot spawn '${agentName}'. Allowed: ${spawnPolicy.allowedErrorText}`, + ); + } +} + +/** + * Resolve every policy shared by task and eval before allocating artifacts or + * dispatching work. Callers translate {@link StructuredSubagentError} into + * their own wire-level error surface. + */ +export async function resolveEffectiveSubagentPolicy( + request: StructuredSubagentRequest, +): Promise { + const spawnPolicy = resolveSpawnPolicy(request.session.getSessionSpawns()); + const agentName = request.agent?.trim() || spawnPolicy.defaultAgent; + const planMode = request.session.getPlanModeState?.()?.enabled === true; + assertPlanControlsAllowed(request, planMode); + assertDepthAndSpawnAllowed(request, agentName); + + const discovery = await discoverAgents(request.session.cwd); + const agent = getAgent(discovery.agents, agentName); + if (!agent) { + const available = discovery.agents.map(candidate => candidate.name).join(", ") || "none"; + throw new StructuredSubagentError("preflight", `Unknown agent "${agentName}". Available: ${available}`); + } + const disabledAgents = request.session.settings.get("task.disabledAgents") as string[]; + if (disabledAgents.includes(agentName)) { + const enabled = discovery.agents + .filter(candidate => !disabledAgents.includes(candidate.name)) + .map(candidate => candidate.name); + throw new StructuredSubagentError( + "preflight", + `Agent "${agentName}" is disabled in settings. Enable it via /agents, or use a different agent type.${enabled.length > 0 ? ` Available: ${enabled.join(", ")}` : ""}`, + ); + } + + const effectiveAgent = planMode ? createPlanModeAgent(agent) : agent; + const schema = resolveSchema(request, effectiveAgent); + if (schema.source === "caller" || (schema.source !== "none" && schema.mode === "strict")) { + const { error } = buildOutputValidator(schema.schema); + if (error) { + const scope = + schema.source === "caller" ? (schema.mode === "strict" ? "strict caller" : "caller") : "strict effective"; + throw new StructuredSubagentError("preflight", `Invalid ${scope} output schema: ${error}`); + } + } + const agentModelOverrides = request.session.settings.get("task.agentModelOverrides"); + const parentActiveModelPattern = request.session.getActiveModelString?.(); + const modelOverride = resolveAgentModelPatterns({ + settingsOverride: request.model ?? agentModelOverrides[agentName], + agentModel: effectiveAgent.model, + settings: request.session.settings, + activeModelPattern: parentActiveModelPattern, + fallbackModelPattern: request.session.getModelString?.(), + }); + const isolationMode = request.session.settings.get("task.isolation.mode"); + const isIsolated = request.isolation?.requested === true; + if (isIsolated && isolationMode === "none") { + throw new StructuredSubagentError( + "preflight", + `Subagent isolated execution requires task.isolation.mode to be set; current mode is "none".`, + ); + } + return { + discovery, + agentName, + agent, + effectiveAgent, + modelOverride, + parentActiveModelPattern, + schema, + planMode, + isIsolated, + mergeMode: request.isolation?.merge ?? request.session.settings.get("task.isolation.merge"), + applyChanges: + request.isolation?.apply ?? + (request.invocationKind === "task" ? request.session.settings.get("task.isolation.apply") : true), + enableLsp: + !planMode && + (request.enableLsp ?? ((request.session.enableLsp ?? true) && request.session.settings.get("task.enableLsp"))), + enableIrc: + !planMode && + (request.enableIrc ?? + (request.session.enableIrc !== false && + isIrcEnabled(request.session.settings, request.session.taskDepth ?? 0))), + }; +} + +/** Reserve a session-global agent id only after preflight has succeeded. */ +export async function reserveStructuredSubagentId( + session: ToolSession, + identity: StructuredSubagentIdentity | undefined, +): Promise { + if (identity?.id) return identity.id; + const manager = session.agentOutputManager ?? new AgentOutputManager(session.getArtifactsDir ?? (() => null)); + session.agentOutputManager ??= manager; + return manager.allocate(sanitizeAgentId(identity?.label) ?? generateTaskName()); +} + +interface ArtifactLease { + sessionFile: string | null; + artifactsDir: string; + temporary: boolean; + unregister: (() => void) | undefined; +} + +async function leaseArtifacts( + session: ToolSession, + invocationKind: StructuredSubagentRequest["invocationKind"], +): Promise { + const sessionFile = session.getSessionFile(); + if (sessionFile) { + const artifactsDir = sessionFile.slice(0, -6); + await fs.mkdir(artifactsDir, { recursive: true }); + return { sessionFile, artifactsDir, temporary: false, unregister: undefined }; + } + const artifactsDir = path.join( + os.tmpdir(), + `${invocationKind === "eval" ? "omp-eval-agent" : "omp-task"}-${Snowflake.next()}`, + ); + await fs.mkdir(artifactsDir, { recursive: true }); + return { sessionFile: null, artifactsDir, temporary: true, unregister: registerArtifactsDir(artifactsDir) }; +} + +function resolveAutoloadSkills(session: ToolSession, agent: AgentDefinition) { + const skills = [...(session.skills ?? [])]; + const autoloadSkills = agent.autoloadSkills?.length + ? agent.autoloadSkills.map(name => skills.find(skill => skill.name === name)).filter(skill => skill !== undefined) + : []; + return { skills, autoloadSkills }; +} + +function buildExecutorOptions( + request: StructuredSubagentRequest, + policy: EffectiveSubagentPolicy, + lease: ArtifactLease, + id: string, +): ExecutorOptions { + const { session } = request; + const { skills, autoloadSkills } = resolveAutoloadSkills(session, policy.agent); + const localProtocolOptions: LocalProtocolOptions = session.localProtocolOptions ?? { + getArtifactsDir: session.getArtifactsDir ?? (() => null), + getSessionId: session.getSessionId ?? (() => null), + }; + const enableMCP = !policy.planMode && (session.enableMCP ?? true); + return { + cwd: session.cwd, + agent: policy.effectiveAgent, + task: renderSubagentPrompt(request.assignment), + assignment: request.assignment.trim(), + context: request.context?.trim() || undefined, + planReference: undefined, + description: trimToUndefined(request.identity?.label), + index: request.index ?? 0, + parentToolCallId: request.parentToolCallId, + detached: request.detached, + id, + taskDepth: session.taskDepth ?? 0, + invokedAt: request.invokedAt, + acquiredAt: request.acquiredAt, + modelOverride: policy.modelOverride, + parentActiveModelPattern: policy.parentActiveModelPattern, + thinkingLevel: policy.effectiveAgent.thinkingLevel, + ...(policy.schema.source === "none" + ? {} + : { + outputSchemaSource: policy.schema.source, + outputSchema: policy.schema.schema, + outputSchemaOverridesAgent: policy.schema.outputSchemaOverridesAgent, + outputSchemaMode: policy.schema.mode, + }), + sessionFile: lease.sessionFile, + persistArtifacts: !lease.temporary, + artifactsDir: lease.artifactsDir, + enableLsp: policy.enableLsp, + enableIrc: policy.enableIrc, + maxRuntimeMs: request.maxRuntimeMs, + restrictToolNames: policy.planMode, + keepAlive: request.keepAlive, + signal: request.signal, + eventBus: session.eventBus, + onProgress: request.onProgress, + authStorage: session.authStorage, + modelRegistry: session.modelRegistry, + settings: session.settings, + mcpManager: enableMCP ? (session.mcpManager ?? MCPManager.instance()) : undefined, + enableMCP, + contextFiles: session.contextFiles?.filter(file => path.basename(file.path).toLowerCase() !== "agents.md"), + skills, + autoloadSkills, + workspaceTree: session.workspaceTree, + promptTemplates: session.promptTemplates, + rules: session.rules, + preloadedExtensionPaths: policy.planMode ? [] : session.extensionPaths, + preloadedCustomToolPaths: policy.planMode ? [] : session.customToolPaths, + localProtocolOptions, + parentArtifactManager: session.getArtifactManager?.() ?? undefined, + parentHindsightSessionState: session.getHindsightSessionState?.(), + parentMnemopiSessionState: session.getMnemopiSessionState?.(), + parentTelemetry: session.getTelemetry?.(), + parentEvalSessionId: request.shareEvalSession === false ? undefined : (session.getEvalSessionId?.() ?? undefined), + parentAgentId: session.getAgentId?.() ?? MAIN_AGENT_ID, + parentServiceTier: session.getServiceTierByFamily ? (session.getServiceTierByFamily() ?? null) : undefined, + }; +} + +async function loadPlanReference( + request: StructuredSubagentRequest, + policy: EffectiveSubagentPolicy, +): Promise<{ path: string; content: string } | undefined> { + if (policy.planMode) return undefined; + const localProtocolOptions: LocalProtocolOptions = request.session.localProtocolOptions ?? { + getArtifactsDir: request.session.getArtifactsDir ?? (() => null), + getSessionId: request.session.getSessionId ?? (() => null), + }; + return loadOverallPlanReference(request.session.getPlanReferencePath?.() ?? "local://PLAN.md", localProtocolOptions); +} + +function buildFailureResult( + request: StructuredSubagentRequest, + policy: EffectiveSubagentPolicy, + id: string, + startedAt: number, +) { + return (error: unknown): SingleResult => { + const message = error instanceof Error ? error.message : String(error); + return { + index: request.index ?? 0, + id, + agent: policy.agent.name, + agentSource: policy.agent.source, + task: renderSubagentPrompt(request.assignment), + assignment: request.assignment.trim(), + description: trimToUndefined(request.identity?.label), + exitCode: 1, + output: "", + stderr: message, + truncated: false, + durationMs: Date.now() - startedAt, + tokens: 0, + requests: 0, + modelOverride: policy.modelOverride, + error: message, + }; + }; +} + +async function persistNestedPatches( + artifactsDir: string, + agentId: string, + nestedPatches: NestedRepoPatch[], +): Promise { + const saved: string[] = []; + for (const [index, nestedPatch] of nestedPatches.entries()) { + const destination = path.join( + artifactsDir, + `${agentId}.nested-${index}-${nestedPatch.relativePath.replace(/[^a-zA-Z0-9._-]/g, "_") || "root"}.patch`, + ); + try { + await fs.writeFile(destination, nestedPatch.patch); + saved.push(destination); + } catch {} + } + return saved; +} + +async function isolationRecoveryHint(result: SingleResult, artifactsDir: string): Promise { + const hints: string[] = []; + if (result.patchPath) hints.push(`Captured patch preserved at ${result.patchPath}.`); + for (const nestedPath of await persistNestedPatches(artifactsDir, result.id, result.nestedPatches ?? [])) { + hints.push(`Captured nested patch preserved at ${nestedPath}.`); + } + if (result.branchName) hints.push(`Captured branch preserved as ${result.branchName}.`); + return hints.length > 0 ? ` ${hints.join(" ")}` : ""; +} + +function attachStructuredOutputMetadata(result: SingleResult, schema: StructuredSubagentSchemaResolution): void { + if (schema.source === "none") { + delete result.structuredOutput; + return; + } + if (result.structuredOutput) return; + let fallbackData: unknown = result.output; + try { + fallbackData = JSON.parse(result.output); + } catch {} + const output: StructuredSubagentOutput = { + source: schema.source, + mode: schema.mode, + status: result.exitCode === 0 ? "valid" : "invalid", + data: fallbackData, + ...(result.error ? { error: result.error } : {}), + }; + result.structuredOutput = output; +} + +/** + * Execute a validated subagent. Preflight errors occur before any artifact + * lease or child dispatch; callers keep responsibility for their result text. + */ +export async function runStructuredSubagent(request: StructuredSubagentRequest): Promise { + const policy = await resolveEffectiveSubagentPolicy(request); + const lease = await leaseArtifacts(request.session, request.invocationKind); + let changesApplied: boolean | null = null; + let mergeSummary = ""; + let requiresRecoveryArtifacts = false; + let completedSuccessfully = false; + try { + const id = await reserveStructuredSubagentId(request.session, { + ...request.identity, + label: request.identity?.label ?? (request.invocationKind === "eval" ? "EvalAgent" : undefined), + }); + const baseOptions = buildExecutorOptions(request, policy, lease, id); + baseOptions.planReference = await loadPlanReference(request, policy); + let isolationContext: IsolationContext | null = null; + if (policy.isIsolated) { + try { + isolationContext = await prepareIsolationContext(request.session.cwd); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); + throw new StructuredSubagentError( + "isolation", + `Isolated subagent execution requires a git repository. ${message}`, + { cause: error }, + ); + } + } + const result = !isolationContext + ? await runSubprocess(baseOptions) + : await runIsolatedSubprocess({ + baseOptions, + context: isolationContext, + preferredBackend: parseIsolationMode(request.session.settings.get("task.isolation.mode")), + agentId: id, + mergeMode: policy.mergeMode, + artifactsDir: lease.artifactsDir, + description: trimToUndefined(request.identity?.label), + buildCommitMessage: makeIsolationCommitMessage(request.session), + buildFailureResult: buildFailureResult(request, policy, id, Date.now()), + }); + attachStructuredOutputMetadata(result, policy.schema); + requiresRecoveryArtifacts = + policy.isIsolated && + (result.exitCode !== 0 || result.error !== undefined || result.aborted === true) && + (result.patchPath !== undefined || result.branchName !== undefined || (result.nestedPatches?.length ?? 0) > 0); + + if ( + policy.isIsolated && + isolationContext && + policy.applyChanges && + result.exitCode === 0 && + !result.error && + !result.aborted + ) { + const outcome = await mergeIsolatedChanges({ + result, + repoRoot: isolationContext.repoRoot, + mergeMode: policy.mergeMode, + }); + mergeSummary = outcome.summary; + changesApplied = outcome.changesApplied; + if (outcome.changesApplied !== false) { + const nestedPatchSummary = await applyEligibleNestedPatches({ + result, + repoRoot: isolationContext.repoRoot, + mergeMode: policy.mergeMode, + changesApplied: outcome.changesApplied, + mergedBranchForNestedPatches: outcome.mergedBranchForNestedPatches, + commitMessage: makeIsolationCommitMessage(request.session)(), + }); + mergeSummary += nestedPatchSummary; + requiresRecoveryArtifacts ||= + nestedPatchSummary.includes("") && (result.nestedPatches?.length ?? 0) > 0; + } + } else if (policy.isIsolated && isolationContext && !policy.applyChanges) { + if (result.branchName) + mergeSummary = `\n\nIsolation: changes captured on branch \`${result.branchName}\` (apply=false). Not merged.`; + else if (result.patchPath) + mergeSummary = `\n\nIsolation: changes captured at \`${result.patchPath}\` (apply=false). Not applied.`; + else if ((result.nestedPatches?.length ?? 0) > 0) + mergeSummary = `\n\nIsolation: changes captured for ${result.nestedPatches?.length} nested ${(result.nestedPatches?.length ?? 0) === 1 ? "repository" : "repositories"} (apply=false). Not applied.`; + else mergeSummary = "\n\nIsolation: no changes captured."; + } + + completedSuccessfully = result.exitCode === 0 && !result.error && !result.aborted; + return { + result, + policy, + mergeSummary, + changesApplied, + artifactsDir: lease.artifactsDir, + temporaryArtifacts: lease.temporary, + }; + } catch (error) { + if (error instanceof StructuredSubagentError) throw error; + throw new StructuredSubagentError( + "execution", + `Subagent execution failed: ${error instanceof Error ? error.message : String(error)}`, + { cause: error }, + ); + } finally { + const shouldRetainArtifacts = + (request.retainArtifacts && completedSuccessfully) || + (policy.isIsolated && (!policy.applyChanges || changesApplied === false || requiresRecoveryArtifacts)); + const shouldCleanup = lease.temporary && !shouldRetainArtifacts; + if (shouldCleanup) { + await fs.rm(lease.artifactsDir, { recursive: true, force: true }); + lease.unregister?.(); + } + } +} + +/** Build the recovery suffix used by adapters after an isolated failure. */ +export async function buildStructuredSubagentRecoveryHint(result: SingleResult, artifactsDir: string): Promise { + return isolationRecoveryHint(result, artifactsDir); +} diff --git a/packages/coding-agent/src/task/types.ts b/packages/coding-agent/src/task/types.ts index 9bba6e785..5a93c9f43 100644 --- a/packages/coding-agent/src/task/types.ts +++ b/packages/coding-agent/src/task/types.ts @@ -7,6 +7,35 @@ import type { NestedRepoPatch } from "./worktree"; /** Source of an agent definition */ export type AgentSource = "bundled" | "user" | "project"; +/** + * Enforcement policy for a structured subagent output schema. + * + * `permissive` preserves legacy retry-budget overrides; `strict` turns every + * invalid final payload, including an exhausted retry override, into a failed + * `schema_violation` result. + */ +export type StructuredSubagentSchemaMode = "permissive" | "strict"; + +/** Origin of the schema selected for a structured subagent invocation. */ +export type StructuredSubagentSchemaSource = "caller" | "agent" | "session" | "none"; + +/** Final validation state of a structured subagent invocation. */ +export type StructuredSubagentValidationStatus = "valid" | "invalid" | "unavailable"; + +/** + * Parsed structured completion and its schema-validation metadata. + * + * `data` is present whenever a payload could be assembled or parsed, even when + * strict validation rejects it. `error` explains unavailable or invalid + * validation without requiring consumers to parse presentation text. + */ +export interface StructuredSubagentOutput { + source: StructuredSubagentSchemaSource; + mode: StructuredSubagentSchemaMode; + status: StructuredSubagentValidationStatus; + data?: unknown; + error?: string; +} const parseNumber = (value: string | undefined, defaultValue: number): number => { if (value) { @@ -77,16 +106,25 @@ export interface SubagentLifecyclePayload { /** Display cap for a normalized one-line label (roster line, registry `displayName`, prompt field). */ export const LABEL_MAX = 80; +// Keep this explicit: ArkType serializes `unknown` as a boolean subschema, which llama.cpp grammars reject. +const outputSchemaInputSchema = type("object | boolean | string | null"); + export const taskItemSchema = type({ "name?": "string", agent: "string = 'task'", task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "+": "delete", }); const taskItemSchemaIsolated = type({ "name?": "string", agent: "string = 'task'", task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "isolated?": "boolean", "+": "delete", }); @@ -99,6 +137,12 @@ export interface TaskItem { agent?: string; /** The work; required by the schema. */ task?: string; + /** Explicit model selector or fallback chain for this spawn, including optional reasoning suffixes. */ + model?: string | string[]; + /** Caller-provided output schema; its presence overrides the selected agent's schema. */ + outputSchema?: unknown; + /** Validation behavior for a caller-provided or inherited output schema. */ + schemaMode?: "permissive" | "strict"; /** Run this spawn in an isolated worktree (batch form; flat form carries it top-level). */ isolated?: boolean; } @@ -107,6 +151,9 @@ export const taskSchema = type({ "name?": "string", agent: "string = 'task'", task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "isolated?": "boolean", "+": "delete", }); @@ -114,6 +161,9 @@ const taskSchemaNoIsolation = type({ "name?": "string", agent: "string = 'task'", task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "+": "delete", }); const taskSchemaBatch = type({ @@ -156,6 +206,9 @@ function createTaskSchema(options: { "name?": "string", agent, task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "isolated?": "boolean", "+": "delete", }); @@ -169,6 +222,9 @@ function createTaskSchema(options: { "name?": "string", agent, task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "+": "delete", }); return type.raw({ @@ -182,6 +238,9 @@ function createTaskSchema(options: { "name?": "string", agent, task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "isolated?": "boolean", "+": "delete", }); @@ -190,6 +249,9 @@ function createTaskSchema(options: { "name?": "string", agent, task: "string", + "model?": "string | string[]", + "outputSchema?": outputSchemaInputSchema, + "schemaMode?": '"permissive" | "strict"', "+": "delete", }); } @@ -231,6 +293,12 @@ export interface TaskParams { agent?: string; /** The work (flat form). */ task?: string; + /** Explicit model selector or fallback chain for the spawn, including optional reasoning suffixes. */ + model?: string | string[]; + /** Caller-provided output schema; its presence overrides the selected agent's schema. */ + outputSchema?: unknown; + /** Validation behavior for a caller-provided or inherited output schema. */ + schemaMode?: "permissive" | "strict"; /** Batch form (`task.batch`): one subagent per item. */ tasks?: TaskItem[]; /** Batch form: shared background prepended to every assignment; required by the batch schema. */ @@ -363,6 +431,8 @@ export interface AgentProgress { modelOverride?: string | string[]; /** Resolved model display string in the form `/`, optionally suffixed with `:` when the level was set explicitly. Undefined when the model could not be resolved. */ resolvedModel?: string; + /** True when {@link resolvedModel} is the target of an active retry fallback (not the originally configured model). Lets observer-only UIs (collab guests, Agent Hub rows with no live session) flag the fallback and keep the provider. */ + resolvedModelIsFallback?: boolean; /** Data extracted by registered subprocess tool handlers (keyed by tool name) */ extractedToolData?: Record; /** @@ -413,6 +483,11 @@ export interface SingleResult { output: string; stderr: string; truncated: boolean; + /** + * Parsed structured completion and validation metadata, when this invocation + * selected an output schema or strict schema mode. + */ + structuredOutput?: StructuredSubagentOutput; durationMs: number; /** Cumulative input + output + cacheWrite tokens across all turns. Excludes cacheRead (re-reads cached context every turn, making cumulative sum misleading). */ tokens: number; @@ -425,6 +500,8 @@ export interface SingleResult { modelOverride?: string | string[]; /** Resolved model display string in the form `/`, optionally suffixed with `:` when the level was set explicitly. Omitted from tool-result JSON when undefined to keep wire payloads small. */ resolvedModel?: string; + /** True when {@link resolvedModel} is the target of an active retry fallback. Mirrors {@link AgentProgress.resolvedModelIsFallback} onto the settled result. */ + resolvedModelIsFallback?: boolean; error?: string; aborted?: boolean; abortReason?: string; diff --git a/packages/coding-agent/src/task/worktree.ts b/packages/coding-agent/src/task/worktree.ts index 86cdf7314..c7e60d8fd 100644 --- a/packages/coding-agent/src/task/worktree.ts +++ b/packages/coding-agent/src/task/worktree.ts @@ -116,7 +116,16 @@ async function captureRepoBaseline(repoRoot: string): Promise { return { repoRoot, headCommit, staged, unstaged, untracked, untrackedPatch }; } -async function writeSyntheticTree(repoDir: string, baseTreeish: string, patches: readonly string[]): Promise { +interface SyntheticTreeOptions { + readonly threeWay?: boolean; +} + +async function writeSyntheticTree( + repoDir: string, + baseTreeish: string, + patches: readonly string[], + options: SyntheticTreeOptions = {}, +): Promise { const tempIndex = path.join(os.tmpdir(), `omp-task-index-${Snowflake.next()}`); try { await git.readTree(repoDir, baseTreeish, { @@ -127,6 +136,7 @@ async function writeSyntheticTree(repoDir: string, baseTreeish: string, patches: await git.patch.applyText(repoDir, patch, { cached: true, env: { GIT_INDEX_FILE: tempIndex }, + threeWay: options.threeWay, }); } return await git.writeTree(repoDir, { @@ -414,6 +424,8 @@ export async function ensureIsolation( preferred?: IsoBackendKind, ): Promise { const repoRoot = await getRepoRoot(baseCwd); + const repository = await git.repo.resolve(repoRoot); + const sourceCommonDir = repository?.commonDir ?? path.join(repoRoot, ".git"); const baseDir = getWorktreeDir(getTaskIsolationSegment(repoRoot, id)); const mergedDir = path.join(baseDir, TASK_ISOLATION_MOUNT_DIR); const resolution = natives.isoResolve(preferred ?? null); @@ -424,6 +436,14 @@ export async function ensureIsolation( await fs.rm(baseDir, { recursive: true, force: true }); try { await natives.isoStart(candidate, repoRoot, mergedDir); + // Sever the isolation's git metadata from the source checkout. Copy + // backends duplicate `repoRoot`'s `.git` verbatim — a linked-worktree + // pointer file (or the rcopy `git worktree add` registration) leaves + // the isolation sharing the source's HEAD/index/ref namespace, so a + // task's git operations would mutate the parent checkout and stack + // parallel task branches. Detaching gives each isolation a private, + // frozen repo that still borrows the source object DB via alternates. + await git.detachGitDir(mergedDir, sourceCommonDir); return { mergedDir, backend: candidate, @@ -643,11 +663,12 @@ async function replayFilteredAgentCommits(opts: FilteredAgentReplayOptions): Pro try { await git.worktree.add(opts.repoRoot, tmpDir, opts.branchName); const agentCommits = await git.revList.range(opts.isolationDir, baselineSha, opts.isolationHead); - const dirtyBaselineTree = await writeSyntheticTree(opts.isolationDir, baselineSha, [ - opts.baseline.root.staged, - opts.baseline.root.unstaged, - opts.baseline.root.untrackedPatch, - ]); + const baselineWip = [opts.baseline.root.staged, opts.baseline.root.unstaged, opts.baseline.root.untrackedPatch]; + // Seed the parent ODB with the dirty-side blobs needed by `git apply + // --3way`. Isolation repositories can read parent objects, but the parent + // cannot read objects created only inside isolation. + await writeSyntheticTree(opts.repoRoot, baselineSha, baselineWip); + const dirtyBaselineTree = await writeSyntheticTree(opts.isolationDir, baselineSha, baselineWip); let previousFilteredTree = baselineSha; let filteredCommitsApplied = 0; @@ -656,7 +677,9 @@ async function replayFilteredAgentCommits(opts: FilteredAgentReplayOptions): Pro allowFailure: true, binary: true, }); - const currentFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [taskStatePatch]); + const currentFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [taskStatePatch], { + threeWay: true, + }); const commitPatch = await git.diff.tree(opts.repoRoot, previousFilteredTree, currentFilteredTree, { allowFailure: true, binary: true, @@ -689,10 +712,11 @@ async function replayFilteredAgentCommits(opts: FilteredAgentReplayOptions): Pro await commitPatchToBranchWorktree(tmpDir, opts.taskId, opts.rootPatch, msg, undefined, opts.baseline.root); } } else { - // A filtered commit landed; tmpDir has advanced past baselineSha and - // previousFilteredTree is HEAD-derived, so writeSyntheticTree + - // leftoverPatch stay HEAD-based and no WIP seed is needed. - const finalFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [opts.rootPatch]); + // A filtered commit landed; reconstruct the final HEAD-derived tree + // with the same dirty-side blobs and 3-way synthesis used above. + const finalFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [opts.rootPatch], { + threeWay: true, + }); const leftoverPatch = await git.diff.tree(opts.repoRoot, previousFilteredTree, finalFilteredTree, { allowFailure: true, binary: true, diff --git a/packages/coding-agent/src/telemetry-export.ts b/packages/coding-agent/src/telemetry-export.ts index 234d43cf3..0dad0ecc7 100644 --- a/packages/coding-agent/src/telemetry-export.ts +++ b/packages/coding-agent/src/telemetry-export.ts @@ -1,144 +1,500 @@ /** - * OTLP trace export bootstrap. + * OTLP telemetry export bootstrap. * * oh-my-pi's agent core (`@oh-my-pi/pi-agent-core`) emits OpenTelemetry GenAI - * spans through the global `@opentelemetry/api` tracer, but only when a - * TracerProvider is registered in the process — otherwise the API returns a - * no-op tracer and the spans are silently dropped. The shipped CLI never - * registered one, so headless / embedded hosts (e.g. an ACP harness that - * spawns `omp` as a child process) had no way to collect omp's internal traces. + * spans through the global `@opentelemetry/api` tracer, and exposes run-level + * callbacks for metrics/log pipelines. This module registers the OTLP/proto + * trace, log, and metric SDK providers when the standard `OTEL_*` endpoint env + * vars are set so `omp` can be observed by any OTLP collector without vendor + * coupling. * - * This module registers a NodeTracerProvider with an OTLP/proto exporter when - * the standard `OTEL_EXPORTER_OTLP_ENDPOINT` (or `..._TRACES_ENDPOINT`) env var - * is set, following the zero-code OTEL env contract: the exporter reads its - * endpoint, headers, and timeout from `OTEL_EXPORTER_OTLP_*` itself. The - * consuming process configures the destination entirely through env; omp stays - * provider-agnostic and ships no vendor coupling. Only the `http/protobuf` - * transport is supported — an `OTEL_EXPORTER_OTLP*_PROTOCOL` of `grpc` or - * `http/json` declines rather than misrouting spans. - * - * The OTLP/proto exporter on the 2.x line is used deliberately: the 1.x line - * deadlocks under Bun — its `req.on('close')` handler fires a spurious failure - * after the success path. `exporter-trace-otlp-proto@0.218` paired with - * `sdk-trace-base@2.7` exports cleanly on Bun. + * Only the `http/protobuf` transport is supported — an + * `OTEL_EXPORTER_OTLP*_PROTOCOL` of `grpc` or `http/json` declines rather than + * misrouting protobuf payloads. The exporter line is pinned to the 0.218/2.7 + * family validated under Bun; the 1.x OTLP line deadlocks when its + * `req.on("close")` handler fires after a successful export. */ +import type { + AgentRunCoverage, + AgentRunSummary, + AgentTelemetryConfig, + AgentTelemetryWarning, + ChatUsageEvent, + ToolStatus, +} from "@oh-my-pi/pi-agent-core"; import { logger, postmortem } from "@oh-my-pi/pi-utils"; -import type * as TraceNode from "@opentelemetry/sdk-trace-node"; +import { + type Attributes, + type AttributeValue, + type Counter, + context, + type Histogram, + type Meter, + metrics, +} from "@opentelemetry/api"; +import { type LogAttributes, logs, type Logger as OtelLogger, SeverityNumber } from "@opentelemetry/api-logs"; +import { AsyncLocalStorageContextManager } from "@opentelemetry/context-async-hooks"; +import { OTLPLogExporter } from "@opentelemetry/exporter-logs-otlp-proto"; +import { OTLPMetricExporter } from "@opentelemetry/exporter-metrics-otlp-proto"; +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto"; +import { resourceFromAttributes } from "@opentelemetry/resources"; +import { BatchLogRecordProcessor, LoggerProvider } from "@opentelemetry/sdk-logs"; +import { MeterProvider, PeriodicExportingMetricReader } from "@opentelemetry/sdk-metrics"; +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base"; +import { NodeTracerProvider } from "@opentelemetry/sdk-trace-node"; /** * Periodic flush interval. A long-lived `omp` process (the ACP server is * spawned once and reused across many turns) would otherwise hold finished - * spans until the batch window elapses or the process exits. + * telemetry until a batch window elapses or the process exits. */ const FLUSH_INTERVAL_MS = 30_000; -let provider: TraceNode.NodeTracerProvider | undefined; +const SERVICE_NAME = "oh-my-pi"; + +type TelemetrySignal = "trace" | "log" | "metric"; +type OtelLogLevel = "none" | logger.LogLevel; + +interface SignalConfig { + readonly trace: boolean; + readonly log: boolean; + readonly metric: boolean; +} + +const LOG_SEVERITY: Record = { + error: SeverityNumber.ERROR, + warn: SeverityNumber.WARN, + info: SeverityNumber.INFO, + debug: SeverityNumber.DEBUG, +}; + +const LOG_LEVEL_WEIGHT: Record = { + error: 0, + warn: 1, + info: 2, + debug: 3, +}; + +const TOOL_STATUSES = ["ok", "error", "skipped", "blocked", "timeout", "aborted"] satisfies readonly ToolStatus[]; + +let traceProvider: NodeTracerProvider | undefined; +let logProvider: LoggerProvider | undefined; +let meterProvider: MeterProvider | undefined; +let metricRecorder: AgentMetricRecorder | undefined; +let otelLogger: OtelLogger | undefined; +let unregisterLogSink: (() => void) | undefined; let initPromise: Promise | undefined; /** - * Whether {@link initTelemetryExport} registered a real provider. The CLI uses - * this to decide whether to switch on the agent loop's telemetry config — there - * is no point emitting spans into a no-op tracer. + * Whether {@link initTelemetryExport} registered any real OTLP signal provider. + * The CLI uses this to decide whether to switch on the agent loop's telemetry + * hooks; metrics and structured logs need those callbacks even when traces are + * disabled. */ export function isTelemetryExportEnabled(): boolean { - return provider !== undefined; + if (traceProvider) return true; + if (logProvider) return true; + if (meterProvider) return true; + return false; } /** - * Register the global TracerProvider + OTLP exporter when an OTLP endpoint is - * configured via env. Idempotent, and a no-op when no endpoint is set (or when - * the OTEL kill-switches are engaged), so it is safe to call unconditionally at - * startup. + * Merge OTLP metrics/log hooks into an existing agent telemetry config. + * + * The caller still owns content-capture policy, cost estimation, and custom + * attributes. This only appends host-level metrics/log forwarding for the + * providers registered by {@link initTelemetryExport}. + */ +export function createTelemetryExportConfig( + config: AgentTelemetryConfig | undefined, +): AgentTelemetryConfig | undefined { + if (!isTelemetryExportEnabled()) return config; + return { + ...config, + onChatUsage: async event => { + await config?.onChatUsage?.(event); + metricRecorder?.recordChatUsage(event); + }, + onRunEnd: (summary, coverage) => { + config?.onRunEnd?.(summary, coverage); + metricRecorder?.recordRun(summary, coverage); + emitRunSummaryLog(summary, coverage); + }, + onTelemetryWarning: warning => { + config?.onTelemetryWarning?.(warning); + emitTelemetryWarningLog(warning); + }, + }; +} + +/** + * Register global trace/log/meter providers when OTLP endpoints are configured + * through env. Idempotent, and a no-op when no signal has an endpoint (or when + * the OTEL kill-switches are engaged), so startup can call it unconditionally. */ export async function initTelemetryExport(): Promise { - if (provider) return; + if (isTelemetryExportEnabled()) return; if (initPromise) return initPromise; - // The OTEL env contract parses booleans and enum lists case-insensitively, so - // OTEL_SDK_DISABLED=TRUE and OTEL_TRACES_EXPORTER=None must also disable export. if (process.env.OTEL_SDK_DISABLED?.trim().toLowerCase() === "true") return; - if (tracesExporterDisabled(process.env.OTEL_TRACES_EXPORTER)) return; - const endpoint = process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT ?? process.env.OTEL_EXPORTER_OTLP_ENDPOINT; - if (!endpoint) return; + const signalConfig = resolveSignalConfig(); + if (!signalConfig.trace && !signalConfig.log && !signalConfig.metric) return; - // We only ship the http/protobuf transport (the line validated on Bun). The - // OTEL contract lets OTEL_EXPORTER_OTLP*_PROTOCOL select grpc / http/json; - // rather than silently send protobuf-over-HTTP to a grpc :4317 port and lose - // every span, decline when an unsupported protocol is requested. - const protocol = (process.env.OTEL_EXPORTER_OTLP_TRACES_PROTOCOL ?? process.env.OTEL_EXPORTER_OTLP_PROTOCOL) - ?.trim() - .toLowerCase(); - if (protocol && protocol !== "http/protobuf") { - logger.warn( - `OTEL trace export disabled: OTEL_EXPORTER_OTLP_PROTOCOL=${protocol} is unsupported (only http/protobuf)`, - ); - return; - } - - initPromise = registerProvider(); + initPromise = registerProviders(signalConfig); return initPromise; } -async function registerProvider(): Promise { - const [ - { AsyncLocalStorageContextManager }, - { OTLPTraceExporter }, - { resourceFromAttributes }, - { BatchSpanProcessor }, - { NodeTracerProvider }, - ] = await Promise.all([ - import("@opentelemetry/context-async-hooks"), - import("@opentelemetry/exporter-trace-otlp-proto"), - import("@opentelemetry/resources"), - import("@opentelemetry/sdk-trace-base"), - import("@opentelemetry/sdk-trace-node"), - ]); - - // The exporter reads endpoint/headers/timeout from OTEL_EXPORTER_OTLP_* itself, - // so there is nothing to thread through here. - const exporter = new OTLPTraceExporter(); - const tracerProvider = new NodeTracerProvider({ - resource: resourceFromAttributes({ - "service.name": process.env.OTEL_SERVICE_NAME ?? "oh-my-pi", - }), - spanProcessors: [new BatchSpanProcessor(exporter)], +async function registerProviders(signalConfig: SignalConfig): Promise { + const resource = resourceFromAttributes({ + "service.name": process.env.OTEL_SERVICE_NAME ?? SERVICE_NAME, }); - // register() installs the global tracer provider and the W3C trace-context + - // baggage propagators; the explicit AsyncLocalStorage context manager keeps - // parent/child span linkage working under Bun. - tracerProvider.register({ contextManager: new AsyncLocalStorageContextManager().enable() }); - provider = tracerProvider; + + if (signalConfig.trace) { + const exporter = new OTLPTraceExporter(); + traceProvider = new NodeTracerProvider({ + resource, + spanProcessors: [new BatchSpanProcessor(exporter)], + }); + traceProvider.register({ contextManager: new AsyncLocalStorageContextManager().enable() }); + } + + if (signalConfig.metric) { + const exporter = new OTLPMetricExporter(); + meterProvider = new MeterProvider({ + resource, + readers: [new PeriodicExportingMetricReader({ exporter })], + }); + metrics.setGlobalMeterProvider(meterProvider); + metricRecorder = new AgentMetricRecorder(metrics.getMeter("@oh-my-pi/pi-coding-agent")); + } + + if (signalConfig.log) { + const exporter = new OTLPLogExporter(); + logProvider = new LoggerProvider({ + resource, + processors: [new BatchLogRecordProcessor({ exporter })], + }); + logs.setGlobalLoggerProvider(logProvider); + otelLogger = logProvider.getLogger("@oh-my-pi/pi-coding-agent"); + unregisterLogSink = logger.registerLogSink(event => { + emitOtelLog( + event.level, + event.message, + logAttributesFromContext(event.context), + "pi.omp.log", + event.timestamp, + ); + }); + } const flushTimer = setInterval(() => { - provider?.forceFlush().catch(() => {}); + flushTelemetryExport().catch(() => {}); }, FLUSH_INTERVAL_MS); flushTimer.unref(); - // Shut down through postmortem rather than a bare signal listener. postmortem - // owns SIGINT/SIGTERM/SIGHUP/exit and quit(), and awaits registered cleanups - // before calling process.exit — so the batch processor's final OTLP export - // completes instead of being cut off mid-flight on the shutdown path. - postmortem.register("otel-trace-export", async () => { + postmortem.register("otel-export", async () => { clearInterval(flushTimer); - await provider?.shutdown(); + unregisterLogSink?.(); + unregisterLogSink = undefined; + const shutdowns: Promise[] = []; + if (traceProvider) shutdowns.push(traceProvider.shutdown()); + if (logProvider) shutdowns.push(logProvider.shutdown()); + if (meterProvider) shutdowns.push(meterProvider.shutdown()); + await Promise.all(shutdowns); }); } -/** - * Parse the `OTEL_TRACES_EXPORTER` selection. The value is a case-insensitive, - * comma-separated list; the literal `none` disables span export entirely. - */ -function tracesExporterDisabled(raw: string | undefined): boolean { - if (!raw) return false; - return raw.split(",").some(entry => entry.trim().toLowerCase() === "none"); +function resolveSignalConfig(): SignalConfig { + const signalConfig: SignalConfig = { + trace: signalEnabled( + "trace", + process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT ?? process.env.OTEL_EXPORTER_OTLP_ENDPOINT, + process.env.OTEL_TRACES_EXPORTER, + process.env.OTEL_EXPORTER_OTLP_TRACES_PROTOCOL ?? process.env.OTEL_EXPORTER_OTLP_PROTOCOL, + ), + log: signalEnabled( + "log", + process.env.OTEL_EXPORTER_OTLP_LOGS_ENDPOINT ?? process.env.OTEL_EXPORTER_OTLP_ENDPOINT, + process.env.OTEL_LOGS_EXPORTER, + process.env.OTEL_EXPORTER_OTLP_LOGS_PROTOCOL ?? process.env.OTEL_EXPORTER_OTLP_PROTOCOL, + ), + metric: signalEnabled( + "metric", + process.env.OTEL_EXPORTER_OTLP_METRICS_ENDPOINT ?? process.env.OTEL_EXPORTER_OTLP_ENDPOINT, + process.env.OTEL_METRICS_EXPORTER, + process.env.OTEL_EXPORTER_OTLP_METRICS_PROTOCOL ?? process.env.OTEL_EXPORTER_OTLP_PROTOCOL, + ), + }; + return signalConfig; +} + +function signalEnabled( + signal: TelemetrySignal, + endpoint: string | undefined, + exporterSelection: string | undefined, + protocolSelection: string | undefined, +): boolean { + if (exporterSelection) { + for (const entry of exporterSelection.split(",")) { + if (entry.trim().toLowerCase() === "none") return false; + } + } + if (!endpoint) return false; + + const protocol = protocolSelection?.trim().toLowerCase(); + if (protocol && protocol !== "http/protobuf") { + logger.warn(`OTEL ${signal} export disabled: OTEL_EXPORTER_OTLP_PROTOCOL=${protocol} is unsupported`, { + supported: "http/protobuf", + }); + return false; + } + return true; +} + +class AgentMetricRecorder { + readonly #tokenUsage: Histogram; + readonly #chatCostUsd: Counter; + readonly #runs: Counter; + readonly #steps: Counter; + readonly #chatCalls: Counter; + readonly #chatDurationMs: Histogram; + readonly #toolCalls: Counter; + readonly #toolDurationMs: Histogram; + readonly #errors: Counter; + + constructor(meter: Meter) { + this.#tokenUsage = meter.createHistogram("gen_ai.client.token.usage", { + description: "Token usage reported by GenAI chat calls.", + unit: "{token}", + }); + this.#chatCostUsd = meter.createCounter("pi.omp.agent.chat.cost.estimated_usd", { + description: "Estimated USD cost for completed chat calls.", + unit: "USD", + }); + this.#runs = meter.createCounter("pi.omp.agent.runs", { + description: "Completed agent runs.", + unit: "{run}", + }); + this.#steps = meter.createCounter("pi.omp.agent.steps", { + description: "Agent loop steps completed inside a run.", + unit: "{step}", + }); + this.#chatCalls = meter.createCounter("pi.omp.agent.chat.calls", { + description: "Chat calls completed inside agent runs.", + unit: "{call}", + }); + this.#chatDurationMs = meter.createHistogram("pi.omp.agent.chat.duration", { + description: "Total chat latency observed in an agent run.", + unit: "ms", + }); + this.#toolCalls = meter.createCounter("pi.omp.agent.tool.calls", { + description: "Tool calls completed inside agent runs.", + unit: "{call}", + }); + this.#toolDurationMs = meter.createHistogram("pi.omp.agent.tool.duration", { + description: "Total tool latency observed in an agent run.", + unit: "ms", + }); + this.#errors = meter.createCounter("pi.omp.agent.errors", { + description: "Errors observed in chat and tool execution.", + unit: "{error}", + }); + } + + recordChatUsage(event: ChatUsageEvent): void { + const baseAttrs = metricAttributes({ + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": event.provider, + "gen_ai.request.model": event.model, + "gen_ai.response.service_tier": event.serviceTier, + "pi.gen_ai.agent.id": event.agent?.id, + "pi.gen_ai.agent.name": event.agent?.name, + }); + + this.#recordToken(event.usage.inputTokens, baseAttrs, "input"); + this.#recordToken(event.usage.outputTokens, baseAttrs, "output"); + this.#recordToken(event.usage.totalTokens, baseAttrs, "total"); + this.#recordToken(event.usage.cachedInputTokens, baseAttrs, "cache_read_input"); + this.#recordToken(event.usage.cacheWriteTokens, baseAttrs, "cache_write_input"); + this.#recordToken(event.usage.reasoningOutputTokens, baseAttrs, "reasoning_output"); + + if (event.cost && "usd" in event.cost && event.cost.usd > 0) { + this.#chatCostUsd.add(event.cost.usd, baseAttrs); + } + } + + recordRun(summary: AgentRunSummary, coverage: AgentRunCoverage): void { + const runAttrs = metricAttributes({ + "pi.omp.agent.models_used.count": coverage.modelsUsed.length, + "pi.omp.agent.providers_used.count": coverage.providersUsed.length, + "pi.omp.agent.tools_available.count": coverage.toolsAvailable.length, + "pi.omp.agent.tools_invoked.count": coverage.toolsInvoked.length, + "pi.omp.agent.tools_unused.count": coverage.toolsUnused.length, + }); + + this.#runs.add(1, runAttrs); + if (summary.stepCount > 0) this.#steps.add(summary.stepCount, runAttrs); + if (summary.chats.totalLatencyMs > 0) this.#chatDurationMs.record(summary.chats.totalLatencyMs, runAttrs); + + for (const reason in summary.chats.byStopReason) { + const count = summary.chats.byStopReason[reason]; + if (count > 0) + this.#chatCalls.add(count, metricAttributes({ ...runAttrs, "gen_ai.response.finish_reason": reason })); + } + for (const toolName in summary.tools.byName) { + const counters = summary.tools.byName[toolName]; + const toolAttrs = metricAttributes({ ...runAttrs, "gen_ai.tool.name": toolName }); + if (counters.totalLatencyMs > 0) this.#toolDurationMs.record(counters.totalLatencyMs, toolAttrs); + for (const status of TOOL_STATUSES) { + const count = counters[status]; + if (count > 0) this.#toolCalls.add(count, metricAttributes({ ...toolAttrs, "pi.omp.tool.status": status })); + } + } + for (const errorType in summary.errors.byType) { + const count = summary.errors.byType[errorType]; + if (count > 0) this.#errors.add(count, metricAttributes({ ...runAttrs, "error.type": errorType })); + } + } + + #recordToken(value: number | undefined, baseAttrs: Attributes, tokenType: string): void { + if (!value || value <= 0) return; + this.#tokenUsage.record(value, metricAttributes({ ...baseAttrs, "gen_ai.token.type": tokenType })); + } +} + +function metricAttributes(fields: Readonly>): Attributes { + const out: Attributes = {}; + for (const key in fields) { + const value = fields[key]; + if (value === undefined || value === null) continue; + if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") { + out[key] = value; + continue; + } + const text = String(value); + if (text.length > 0) out[key] = text; + } + return out; +} + +function emitRunSummaryLog(summary: AgentRunSummary, coverage: AgentRunCoverage): void { + emitOtelLog( + "info", + "agent run completed", + { + "pi.omp.agent.step_count": summary.stepCount, + "pi.omp.agent.chats.total": summary.chats.total, + "pi.omp.agent.chats.total_latency_ms": summary.chats.totalLatencyMs, + "pi.omp.agent.tools.total": summary.tools.total, + "pi.omp.agent.tools.ok": summary.tools.ok, + "pi.omp.agent.tools.error": summary.tools.error, + "pi.omp.agent.tools.skipped": summary.tools.skipped, + "pi.omp.agent.tools.blocked": summary.tools.blocked, + "pi.omp.agent.tools.timeout": summary.tools.timeout, + "pi.omp.agent.tools.aborted": summary.tools.aborted, + "pi.omp.agent.tools.total_latency_ms": summary.tools.totalLatencyMs, + "pi.omp.agent.usage.input_tokens": summary.usage.inputTokens, + "pi.omp.agent.usage.output_tokens": summary.usage.outputTokens, + "pi.omp.agent.usage.cached_input_tokens": summary.usage.cachedInputTokens, + "pi.omp.agent.usage.cache_write_tokens": summary.usage.cacheWriteTokens, + "pi.omp.agent.usage.reasoning_output_tokens": summary.usage.reasoningOutputTokens, + "pi.omp.agent.usage.total_tokens": summary.usage.totalTokens, + "pi.omp.agent.cost.estimated_usd": summary.cost.estimatedUsd, + "pi.omp.agent.cost.unavailable_reasons": summary.cost.unavailableReasons.join(","), + "pi.omp.agent.errors.total": summary.errors.total, + "pi.omp.agent.coverage.tools_available": coverage.toolsAvailable.join(","), + "pi.omp.agent.coverage.tools_invoked": coverage.toolsInvoked.join(","), + "pi.omp.agent.coverage.tools_unused": coverage.toolsUnused.join(","), + "pi.omp.agent.coverage.models_used": coverage.modelsUsed.join(","), + "pi.omp.agent.coverage.providers_used": coverage.providersUsed.join(","), + }, + "pi.omp.agent.run.completed", + ); +} + +function emitTelemetryWarningLog(warning: AgentTelemetryWarning): void { + const attrs = logAttributesFromContext({ + code: warning.code, + error: warning.error, + }); + emitOtelLog("warn", warning.message, attrs, "pi.omp.telemetry.warning"); +} + +function emitOtelLog( + level: logger.LogLevel, + body: string, + attributes: LogAttributes, + eventName: string, + timestamp = new Date(), +): void { + if (!otelLogger) return; + const minLevel = parseOtelLogLevel(process.env.OTEL_LOG_LEVEL); + if (minLevel === "none") return; + if (LOG_LEVEL_WEIGHT[level] > LOG_LEVEL_WEIGHT[minLevel]) return; + otelLogger.emit({ + eventName, + timestamp, + observedTimestamp: new Date(), + severityNumber: LOG_SEVERITY[level], + severityText: level.toUpperCase(), + body, + attributes, + context: context.active(), + }); +} + +function parseOtelLogLevel(raw: string | undefined): OtelLogLevel { + if (!raw) return "info"; + switch (raw.trim().toLowerCase()) { + case "none": + return "none"; + case "error": + return "error"; + case "warn": + case "warning": + return "warn"; + case "debug": + return "debug"; + default: + return "info"; + } +} + +function logAttributesFromContext(input: Record | undefined): LogAttributes { + const out: LogAttributes = { "process.pid": process.pid }; + if (!input) return out; + for (const key in input) { + const attr = logAttributeValue(input[key]); + if (attr !== undefined) out[key] = attr; + } + return out; +} + +function logAttributeValue(value: unknown): AttributeValue | undefined { + if (value === undefined || value === null) return undefined; + if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") return value; + if (value instanceof Error) { + return `${value.name}: ${value.message}`; + } + try { + const text = JSON.stringify(value); + if (text && text.length > 0) return text; + } catch { + return String(value); + } + return String(value); } /** - * Flush any buffered spans to the exporter. No-op when export is disabled. + * Flush buffered spans, log records, and metrics. No-op when export is disabled. * Hosts embedding the agent can call this at natural boundaries (e.g. the end - * of a turn) so traces surface promptly rather than on the batch interval. + * of a turn) so telemetry surfaces promptly rather than on the batch interval. */ export async function flushTelemetryExport(): Promise { - await provider?.forceFlush(); + const flushes: Promise[] = []; + if (traceProvider) flushes.push(traceProvider.forceFlush()); + if (logProvider) flushes.push(logProvider.forceFlush()); + if (meterProvider) flushes.push(meterProvider.forceFlush()); + await Promise.all(flushes); } diff --git a/packages/coding-agent/src/tools/__tests__/eval-description.test.ts b/packages/coding-agent/src/tools/__tests__/eval-description.test.ts index 49ff35477..d62140cb1 100644 --- a/packages/coding-agent/src/tools/__tests__/eval-description.test.ts +++ b/packages/coding-agent/src/tools/__tests__/eval-description.test.ts @@ -6,7 +6,7 @@ describe("eval tool description", () => { const description = getEvalToolDescription({ py: true, js: false, spawns: "fact-finder,oracle" }); expect(description).toContain('agent(prompt, agent?="fact-finder"'); - expect(description).toContain("Allowed: `fact-finder`, `oracle`."); + expect(description).toContain("Allowed agents: `fact-finder`, `oracle`."); }); it("omits agent() when spawning is disabled", () => { diff --git a/packages/coding-agent/src/tools/approval.ts b/packages/coding-agent/src/tools/approval.ts index c1d53fbe8..9b39eb0a7 100644 --- a/packages/coding-agent/src/tools/approval.ts +++ b/packages/coding-agent/src/tools/approval.ts @@ -75,6 +75,17 @@ function getToolDecision(tool: ApprovalSubject, args: unknown): Omit | undefined): Record | undefined { if (!env || Object.keys(env).length === 0) return undefined; const normalized: Record = {}; @@ -314,10 +312,17 @@ function extractPartialBashEnv(partialJson: string | undefined): Record 0 ? env : undefined; } -function formatTimeoutClampNotice(requestedTimeoutSec: number, effectiveTimeoutSec: number): string | undefined { - return requestedTimeoutSec !== effectiveTimeoutSec - ? `Timeout clamped to ${effectiveTimeoutSec}s (requested ${requestedTimeoutSec}s; allowed range ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}s).` - : undefined; +function formatTimeoutClampNotice( + requestedTimeoutSec: number, + effectiveTimeoutSec: number, + maxTimeout: number, +): string | undefined { + if (requestedTimeoutSec === effectiveTimeoutSec) return undefined; + const cappedByGlobal = maxTimeout > 0 && effectiveTimeoutSec === maxTimeout && maxTimeout < TOOL_TIMEOUTS.bash.max; + const limit = cappedByGlobal + ? `global tools.maxTimeout ceiling ${maxTimeout}s` + : `allowed range ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}s`; + return `Timeout clamped to ${effectiveTimeoutSec}s (requested ${requestedTimeoutSec}s; ${limit}).`; } function formatWallTimeSeconds(wallTimeMs: number): string { @@ -439,12 +444,15 @@ export class BashTool implements AgentTool saveBashOriginalArtifact(this.session, full), + }); + return toolResult(details) + .text(timeoutOutputText) + .truncationFromSummary(result, { direction: "tail" }) + .error() + .done(); + } + + // Non-timeout cancellations and missing exit status still propagate as thrown errors. + this.#throwIfUnfinished(result, timeoutSec, outputText); + // Final defense at the tool-result boundary: no bash path (client bridge, // head-retention spill, minimizer miss) may emit more than // ~DEFAULT_MAX_BYTES inline. No-op for already-bounded output. @@ -802,11 +838,12 @@ export class BashTool implements AgentTool 0 ? current.output.split("\n").length : 0, outputBytes: current.output.length, }; - return this.#buildCompletedResult(timedOutResult, timeoutSec, { - requestedTimeoutSec, - notices: pendingNotices, - terminalId: handle.terminalId, - wallTimeMs: performance.now() - bridgeWallTimeStart, - }); + this.#throwIfUnfinished(timedOutResult, timeoutSec, this.#formatResultOutput(timedOutResult)); + throw new ToolError("Command timed out"); } if (raced.kind === "exit") { @@ -1102,20 +1135,27 @@ export class BashTool implements AgentTool(config: ShellRendererConfig) { const isError = result.isError === true; const isPartial = options.isPartial === true; const success = !isPartial && !isError; + const details = result.details; + const isTimeout = details?.timedOut === true; const header = config.showHeader === false ? undefined @@ -1263,12 +1305,11 @@ export function createShellRenderer(config: ShellRendererConfig) { title: config.resolveTitle(args, options), } : { - icon: isPartial ? "pending" : "error", + icon: isPartial ? "pending" : isTimeout ? "warning" : "error", title: config.resolveTitle(args, options), }, uiTheme, ); - const details = result.details; const outputBlock = new CachedOutputBlock(); // Per-instance cache for the expensive inner lines computation. Mirrors @@ -1408,7 +1449,7 @@ export function createShellRenderer(config: ShellRendererConfig) { const framed = outputBlock.render( { header, - state: isPartial ? "pending" : isError ? "error" : "success", + state: isPartial ? "pending" : isError ? (isTimeout ? "warning" : "error") : "success", sections: [ { // Viewport-sized tail window in every state — streaming and final diff --git a/packages/coding-agent/src/tools/browser.ts b/packages/coding-agent/src/tools/browser.ts index 0f7d9e093..362d24cd6 100644 --- a/packages/coding-agent/src/tools/browser.ts +++ b/packages/coding-agent/src/tools/browser.ts @@ -190,7 +190,7 @@ export class BrowserTool implements AgentTool> { try { throwIfAborted(signal); - const timeoutSeconds = clampTimeout("browser", params.timeout); + const timeoutSeconds = clampTimeout("browser", params.timeout, this.session.settings.get("tools.maxTimeout")); const timeoutMs = timeoutSeconds * 1000; const name = params.name ?? DEFAULT_TAB_NAME; const details: BrowserToolDetails = { action: params.action, name }; @@ -199,7 +199,7 @@ export class BrowserTool implements AgentTool> { const kill = !!params.kill; if (params.all) { - const count = await untilAborted(signal, () => releaseAllTabs({ kill })); + const count = await untilAborted(signal, () => releaseAllTabs({ kill, timeoutMs })); details.result = `Closed ${count} tab(s)`; return toolResult(details).text(details.result).done(); } - const closed = await untilAborted(signal, () => releaseTab(name, { kill })); + const closed = await untilAborted(signal, () => releaseTab(name, { kill, timeoutMs })); details.result = closed ? `Closed tab ${JSON.stringify(name)}` : `No tab named ${JSON.stringify(name)}`; return toolResult(details).text(details.result).done(); } diff --git a/packages/coding-agent/src/tools/browser/aria/aria-snapshot.ts b/packages/coding-agent/src/tools/browser/aria/aria-snapshot.ts index c37e30983..693251281 100644 --- a/packages/coding-agent/src/tools/browser/aria/aria-snapshot.ts +++ b/packages/coding-agent/src/tools/browser/aria/aria-snapshot.ts @@ -1,4 +1,5 @@ import type { ElementHandle, JSHandle, Page } from "puppeteer-core"; +import { ToolError } from "../../tool-errors"; import ariaBundle from "./aria-snapshot.bundle.txt" with { type: "text" }; // `aria-snapshot.bundle.txt` is a generated, committed artifact: Playwright's // injected ARIA-snapshot sources (pinned, Apache-2.0) bundled to a CJS module. @@ -69,15 +70,41 @@ export async function resolveAriaRefHandle(page: Page, ref: string): Promise; +}; + +function parseRelayCredentials(relayIdValue: unknown, relayTokenValue: unknown): RelayCredentials | null { + if (typeof relayIdValue !== "string" || typeof relayTokenValue !== "string") { + return null; + } + const relayId = relayIdValue.trim(); + const relayTokenHex = relayTokenValue.trim(); + if ( + relayId.length === 0 || + relayTokenHex.length === 0 || + relayTokenHex.length % 2 !== 0 || + !/^[0-9a-f]+$/i.test(relayTokenHex) + ) { + return null; + } + const relayToken = new Uint8Array(new ArrayBuffer(relayTokenHex.length / 2)); + for (let index = 0; index < relayToken.length; index++) { + relayToken[index] = Number.parseInt(relayTokenHex.slice(index * 2, index * 2 + 2), 16); + } + return { relayId, relayToken }; +} + export function formatCmuxError(error: CmuxErrorPayload | undefined): string { const code = typeof error?.code === "string" && error.code.length > 0 ? error.code : "error"; const message = typeof error?.message === "string" && error.message.length > 0 ? error.message : "cmux error"; @@ -35,6 +69,8 @@ export function formatCmuxError(error: CmuxErrorPayload | undefined): string { export class CmuxSocketClient { readonly #socketPath: string; readonly #password: string | undefined; + readonly #relayId: string | undefined; + readonly #relayToken: string | undefined; #socket: net.Socket | null = null; #connectPromise: Promise | null = null; #connected = false; @@ -45,9 +81,11 @@ export class CmuxSocketClient { #activeJob: RequestJob | null = null; #pumping = false; - constructor(opts: { socketPath: string; password?: string }) { + constructor(opts: { socketPath: string; password?: string; relayId?: string; relayToken?: string }) { this.#socketPath = opts.socketPath; this.#password = opts.password; + this.#relayId = opts.relayId ?? process.env.CMUX_RELAY_ID; + this.#relayToken = opts.relayToken ?? process.env.CMUX_RELAY_TOKEN; } async connect(): Promise { @@ -101,7 +139,11 @@ export class CmuxSocketClient { } async #openSocket(): Promise { - const socket = net.createConnection({ path: this.#socketPath }); + const relayEndpoint = this.#parseRelayEndpoint(); + const relayCredentials = relayEndpoint ? await this.#loadRelayCredentials(relayEndpoint) : null; + const socket = relayEndpoint + ? net.createConnection({ host: relayEndpoint.host, port: relayEndpoint.port }) + : net.createConnection({ path: this.#socketPath }); this.#socket = socket; this.#buffer = ""; socket.setEncoding("utf8"); @@ -111,13 +153,16 @@ export class CmuxSocketClient { try { await this.#waitForConnect(socket); - this.#connected = true; + if (relayEndpoint && relayCredentials) { + await this.#authenticateRelay(relayEndpoint, relayCredentials); + } if (this.#password) { const line = await this.#sendLine(`auth ${this.#password}`, DEFAULT_CONNECT_TIMEOUT_MS); if (line.startsWith("ERROR:") && !line.includes("Unknown command 'auth'")) { throw new ToolError(line); } } + this.#connected = true; } catch (err) { this.#connected = false; socket.destroy(); @@ -128,6 +173,97 @@ export class CmuxSocketClient { } } + #parseRelayEndpoint(): RelayEndpoint | null { + const value = this.#socketPath.trim(); + if (value.length === 0 || value.startsWith("/")) { + return null; + } + const match = /^(127\.0\.0\.1|localhost):([0-9]+)$/.exec(value); + if (!match) { + return null; + } + const port = Number.parseInt(match[2] ?? "", 10); + if (!Number.isInteger(port) || port < 1 || port > 65_535) { + return null; + } + return { host: "127.0.0.1", port }; + } + + async #loadRelayCredentials(endpoint: RelayEndpoint): Promise { + const environmentCredentials = parseRelayCredentials(this.#relayId, this.#relayToken); + if (environmentCredentials) { + return environmentCredentials; + } + + const authPath = path.join(os.homedir(), ".cmux", "relay", `${endpoint.port}.auth`); + let payload: unknown; + try { + payload = await Bun.file(authPath).json(); + } catch { + throw new ToolError( + `Missing cmux relay auth metadata for ${endpoint.host}:${endpoint.port}; set CMUX_RELAY_ID/CMUX_RELAY_TOKEN or restore ~/.cmux/relay/${endpoint.port}.auth`, + ); + } + const relayId = payload && typeof payload === "object" && "relay_id" in payload ? payload.relay_id : undefined; + const relayToken = + payload && typeof payload === "object" && "relay_token" in payload ? payload.relay_token : undefined; + const fileCredentials = parseRelayCredentials(relayId, relayToken); + if (!fileCredentials) { + throw new ToolError(`Invalid cmux relay auth metadata in ~/.cmux/relay/${endpoint.port}.auth`); + } + return fileCredentials; + } + + async #authenticateRelay(endpoint: RelayEndpoint, credentials: RelayCredentials): Promise { + const challengeLine = await this.#nextLine(DEFAULT_CONNECT_TIMEOUT_MS); + let challenge: unknown; + try { + challenge = JSON.parse(challengeLine); + } catch { + throw new ToolError(`Invalid cmux relay authentication challenge from ${endpoint.host}:${endpoint.port}`); + } + if ( + !challenge || + typeof challenge !== "object" || + !("protocol" in challenge) || + challenge.protocol !== "cmux-relay-auth" || + !("version" in challenge) || + typeof challenge.version !== "number" || + !Number.isInteger(challenge.version) || + !("relay_id" in challenge) || + challenge.relay_id !== credentials.relayId || + !("nonce" in challenge) || + typeof challenge.nonce !== "string" || + challenge.nonce.length === 0 + ) { + throw new ToolError(`Invalid cmux relay authentication challenge from ${endpoint.host}:${endpoint.port}`); + } + + const message = `relay_id=${challenge.relay_id}\nnonce=${challenge.nonce}\nversion=${challenge.version}`; + const key = await globalThis.crypto.subtle.importKey( + "raw", + credentials.relayToken, + { name: "HMAC", hash: "SHA-256" }, + false, + ["sign"], + ); + const mac = await globalThis.crypto.subtle.sign("HMAC", key, UTF8.encode(message)); + const authLine = JSON.stringify({ + relay_id: credentials.relayId, + mac: Buffer.from(mac).toString("hex"), + }); + const responseLine = await this.#sendLine(authLine, DEFAULT_CONNECT_TIMEOUT_MS); + let response: unknown; + try { + response = JSON.parse(responseLine); + } catch { + throw new ToolError(`Cmux relay authentication failed for ${endpoint.host}:${endpoint.port}`); + } + if (!response || typeof response !== "object" || !("ok" in response) || response.ok !== true) { + throw new ToolError(`Cmux relay authentication failed for ${endpoint.host}:${endpoint.port}`); + } + } + #waitForConnect(socket: net.Socket): Promise { const { promise, resolve, reject } = Promise.withResolvers(); const timer = setTimeout(() => { diff --git a/packages/coding-agent/src/tools/browser/registry.ts b/packages/coding-agent/src/tools/browser/registry.ts index 58dcb7531..a34b63545 100644 --- a/packages/coding-agent/src/tools/browser/registry.ts +++ b/packages/coding-agent/src/tools/browser/registry.ts @@ -47,6 +47,13 @@ export interface CmuxBrowserHandle extends BrowserHandleCommon { export type BrowserHandle = PuppeteerBrowserHandle | CmuxBrowserHandle; +/** Controls bounded browser-handle teardown and identifies the owning resource in timeout diagnostics. */ +export interface ReleaseBrowserOptions { + kill: boolean; + timeoutMs?: number; + resource?: string; +} + const browsers = new Map(); function browserKey(kind: BrowserKind): string { @@ -221,7 +228,7 @@ export function holdBrowser(handle: BrowserHandle): void { handle.refCount++; } -export async function releaseBrowser(handle: BrowserHandle, opts: { kill: boolean }): Promise { +export async function releaseBrowser(handle: BrowserHandle, opts: ReleaseBrowserOptions): Promise { handle.refCount = Math.max(0, handle.refCount - 1); if (handle.refCount === 0) { // Only evict if the registry still points at THIS handle. After a disconnect, @@ -232,7 +239,7 @@ export async function releaseBrowser(handle: BrowserHandle, opts: { kill: boolea } } -async function disposeBrowserHandle(handle: BrowserHandle, opts: { kill: boolean }): Promise { +async function disposeBrowserHandle(handle: BrowserHandle, opts: ReleaseBrowserOptions): Promise { if ("client" in handle) { handle.client.close(); return; diff --git a/packages/coding-agent/src/tools/browser/run-cancellation.ts b/packages/coding-agent/src/tools/browser/run-cancellation.ts index d2313af14..2fc638fce 100644 --- a/packages/coding-agent/src/tools/browser/run-cancellation.ts +++ b/packages/coding-agent/src/tools/browser/run-cancellation.ts @@ -117,6 +117,10 @@ export function bindBrowserRunFacade(target: T, signal: AbortS return wrapped; } if (value && typeof value === "object") { + // Never proxy AbortSignals: native combinators (AbortSignal.any, fetch) + // brand-check internal slots that a Proxy cannot forward, and reading a + // signal needs no abort gating anyway. + if (value instanceof AbortSignal) return value; const wrapped = bindBrowserRunFacade(value, signal); cache.set(prop, wrapped); return wrapped; diff --git a/packages/coding-agent/src/tools/browser/tab-protocol.ts b/packages/coding-agent/src/tools/browser/tab-protocol.ts index 96c946c4a..cd0d6de9a 100644 --- a/packages/coding-agent/src/tools/browser/tab-protocol.ts +++ b/packages/coding-agent/src/tools/browser/tab-protocol.ts @@ -95,6 +95,8 @@ export interface RunErrorPayload { stack?: string; isToolError: boolean; isAbort: boolean; + /** The worker could not restore tab-scoped browser state and must be recycled. */ + recoverTab?: boolean; } export type WorkerOutbound = diff --git a/packages/coding-agent/src/tools/browser/tab-supervisor.ts b/packages/coding-agent/src/tools/browser/tab-supervisor.ts index 0d0a7324a..3a1cc875e 100644 --- a/packages/coding-agent/src/tools/browser/tab-supervisor.ts +++ b/packages/coding-agent/src/tools/browser/tab-supervisor.ts @@ -1,4 +1,4 @@ -import { getPuppeteerDir, logger, postmortem, Snowflake, workerHostEntry } from "@oh-my-pi/pi-utils"; +import { getPuppeteerDir, logger, postmortem, Snowflake, withTimeout, workerHostEntry } from "@oh-my-pi/pi-utils"; import type { Page, Target } from "puppeteer-core"; import { callSessionTool } from "../../eval/js/tool-bridge"; import { webpExclusionForModel } from "../../utils/image-loading"; @@ -123,6 +123,8 @@ export interface RunInTabOptions { export interface ReleaseTabOptions { kill?: boolean; + /** Maximum time for each asynchronous cleanup resource before close fails with diagnostics. */ + timeoutMs?: number; } const tabs = new Map(); @@ -135,6 +137,23 @@ const GRACE_MS = 750; // mapped to the kill reason. Lets the next `run` on that name explain WHY the tab // vanished instead of a bare "not alive". Cleared when the name is opened again. const killedTabs = new Map(); +const DEFAULT_TAB_CLOSE_TIMEOUT_MS = 5_000; +class RecoverableWorkerError extends ToolError {} + +async function waitForTabCleanup( + tab: TabSession, + timeoutMs: number, + pendingResource: string, + promise: Promise, +): Promise { + const message = `Timed out after ${timeoutMs}ms closing ${tab.kindTag} browser tab ${JSON.stringify(tab.name)}; pending resource: ${pendingResource}`; + try { + return await withTimeout(promise, timeoutMs, message); + } catch (error) { + if (error instanceof Error && error.message === message) throw new ToolError(message); + throw error; + } +} export function getTab(name: string): TabSession | undefined { return tabs.get(name); @@ -470,16 +489,23 @@ async function runInTabWithSnapshot( async reason => await forceKillTab(name, reason), ); } catch (error) { - if (error instanceof ToolError && error.message.startsWith("Browser code execution timed out after ")) { + const runTimedOut = + error instanceof ToolError && error.message.startsWith("Browser code execution timed out after "); + if (runTimedOut || error instanceof RecoverableWorkerError) { try { - if (tab.worker.mode === "inline") - await forceKillTab(name, "Browser code execution timed out; tab killed"); - else await recycleTimedOutWorkerTab(tab, opts.timeoutMs + GRACE_MS); + if (tab.worker.mode === "inline") { + const reason = runTimedOut + ? "Browser code execution timed out; tab killed" + : "Browser request interception cleanup failed; tab killed"; + await forceKillTab(name, reason); + } else { + await recycleTimedOutWorkerTab(tab, opts.timeoutMs + GRACE_MS); + } } catch (recycleError) { - logger.warn("Failed to recycle timed-out browser tab worker; killing tab", { + logger.warn("Failed to recycle browser tab worker; killing tab", { error: recycleError instanceof Error ? recycleError.message : String(recycleError), }); - await forceKillTab(name, "Browser code execution timed out; tab killed"); + await forceKillTab(name, "Browser tab worker recovery failed; tab killed"); } } throw error; @@ -519,26 +545,42 @@ export async function releaseTab(name: string, opts: ReleaseTabOptions = {}): Pr pending.reject(closeError); } tab.pending.clear(); + const timeoutMs = opts.timeoutMs ?? DEFAULT_TAB_CLOSE_TIMEOUT_MS; if (tab.backend === "cmux") { - let nonLastCloseError: unknown; + let closeError: unknown; if (wasAlive && tab.cmuxOwnsSurface) { try { - await tab.browser.client.request("surface.close", { surface_id: tab.targetId }); + await waitForTabCleanup( + tab, + timeoutMs, + `cmux surface ${JSON.stringify(tab.targetId)} (surface.close)`, + tab.browser.client.request("surface.close", { surface_id: tab.targetId }, { timeoutMs }), + ); } catch (err) { if (isLastSurfaceCloseError(err)) { logger.debug("Leaving cmux browser surface open because it is the last surface in the workspace", { error: err instanceof Error ? err.message : String(err), }); } else { - nonLastCloseError = err; + closeError = err; } } } - await releaseBrowser(tab.browser, { kill: opts.kill ?? false }); - tabs.delete(name); - if (nonLastCloseError) throw nonLastCloseError; + try { + await releaseBrowser(tab.browser, { + kill: opts.kill ?? false, + timeoutMs, + resource: `tab ${JSON.stringify(name)}`, + }); + } catch (error) { + closeError ??= error; + } finally { + tabs.delete(name); + } + if (closeError) throw closeError; return true; } + let cleanupError: unknown; let forced = false; if (wasAlive) { try { @@ -549,9 +591,30 @@ export async function releaseTab(name: string, opts: ReleaseTabOptions = {}): Pr } } await tab.worker.terminate().catch(() => undefined); - if (forced && tab.kindTag === "headless") await closeOrphanTarget(tab); - await releaseBrowser(tab.browser, { kill: opts.kill ?? false }); - tabs.delete(name); + if (forced && tab.kindTag === "headless") { + try { + await waitForTabCleanup( + tab, + timeoutMs, + `orphan CDP target ${JSON.stringify(tab.targetId)} (Page.close)`, + closeOrphanTarget(tab), + ); + } catch (error) { + cleanupError = error; + } + } + try { + await releaseBrowser(tab.browser, { + kill: opts.kill ?? false, + timeoutMs, + resource: `tab ${JSON.stringify(name)}`, + }); + } catch (error) { + cleanupError ??= error; + } finally { + tabs.delete(name); + } + if (cleanupError) throw cleanupError; return true; } @@ -818,11 +881,13 @@ async function targetIdForTarget(target: Target): Promise { } function errorFromPayload(payload: RunErrorPayload): Error { - const error = payload.isAbort - ? new ToolAbortError() - : payload.isToolError - ? new ToolError(payload.message) - : new Error(payload.message); + const error = payload.recoverTab + ? new RecoverableWorkerError(payload.message) + : payload.isAbort + ? new ToolAbortError() + : payload.isToolError + ? new ToolError(payload.message) + : new Error(payload.message); error.name = payload.name; if (payload.stack) error.stack = payload.stack; return error; diff --git a/packages/coding-agent/src/tools/browser/tab-worker.ts b/packages/coding-agent/src/tools/browser/tab-worker.ts index 55bb31eb2..462c3da78 100644 --- a/packages/coding-agent/src/tools/browser/tab-worker.ts +++ b/packages/coding-agent/src/tools/browser/tab-worker.ts @@ -2,7 +2,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { postmortem, Snowflake, untilAborted } from "@oh-my-pi/pi-utils"; +import { postmortem, Snowflake, untilAborted, withTimeout } from "@oh-my-pi/pi-utils"; import type { HTMLElement } from "linkedom"; import type { Browser, @@ -24,6 +24,7 @@ import { formatScreenshot } from "../render-utils"; import { ToolAbortError, ToolError, throwIfAborted } from "../tool-errors"; import { type AriaSnapshotOptions, + assertSelectorString, captureAriaSnapshot, parseAriaRefSelector, resolveAriaRefHandle, @@ -37,6 +38,7 @@ import { } from "./launch"; import { extractReadableFromHtml, type ReadableFormat } from "./readable"; import { + bindBrowserRunFacade, CELL_BUDGET_SLACK_MS, markHandled, resolvePredicateTimeout, @@ -144,6 +146,8 @@ interface OpenDialogInfo { */ const QUICK_OP_TIMEOUT_MS = 20_000; const ACTION_OP_TIMEOUT_MS = 8_000; +/** Maximum wait for a renderer acknowledgement after a wheel event is queued. */ +const SCROLL_ACK_TIMEOUT_MS = 2_000; /** Headroom subtracted from the cell budget so a per-op deadline fires before it. */ const OP_DEADLINE_SLACK_MS = CELL_BUDGET_SLACK_MS; /** @@ -155,6 +159,8 @@ const OP_DEADLINE_SLACK_MS = CELL_BUDGET_SLACK_MS; const ZERO_MATCH_FAIL_FAST_MS = 2_000; /** Poll cadence for the zero-match watchdog. */ const ZERO_MATCH_POLL_MS = 250; +/** Cleanup must settle inside the supervisor's 750ms post-run grace window. */ +const REQUEST_INTERCEPTION_CLEANUP_TIMEOUT_MS = 500; export interface OpTimeouts { /** Largest per-op deadline allowed — strictly below the cell budget. */ @@ -175,6 +181,21 @@ export function resolveOpTimeouts(cellTimeoutMs: number): OpTimeouts { }; } +/** Queue a wheel event without treating a delayed renderer acknowledgement as dispatch failure. */ +export async function dispatchScroll( + dispatch: () => Promise, + ackTimeoutMs = SCROLL_ACK_TIMEOUT_MS, +): Promise { + const deadline = Promise.withResolvers(); + const timer = setTimeout(() => deadline.resolve(), ackTimeoutMs); + timer.unref(); + try { + await Promise.race([dispatch(), deadline.promise]); + } finally { + clearTimeout(timer); + } +} + /** * Effective timeout for a wait helper (`waitFor*`). A positive explicit `{ timeout }` is * honored but clamped to the cell budget so it still fails fast + named; raising the tool @@ -246,6 +267,7 @@ interface TabApi { } export function normalizeSelector(selector: string): string { + assertSelectorString(selector); if (!selector) return selector; if ( !SELECTOR_HANDLER_PREFIXES.some(prefix => selector.startsWith(prefix)) && @@ -334,12 +356,125 @@ function redactUrlCredentials(url: string): string { } } +class RequestInterceptionCleanupError extends ToolError {} + +interface RunPageScope { + page: Page; + cleanup(): Promise; +} + +/** + * Expose the tab page while retaining the request handlers created by this run. + * Puppeteer's Page wraps an internal emitter, so `removeAllListeners("request")` + * would also remove its forwarding listener; the facade removes only user handlers. + */ +function createRunPageScope(page: Page): RunPageScope { + const requestHandlers: unknown[] = []; + const on = page.on; + const off = page.off; + const once = page.once; + const removeAllListeners = page.removeAllListeners; + const onDescriptor = Object.getOwnPropertyDescriptor(page, "on"); + const offDescriptor = Object.getOwnPropertyDescriptor(page, "off"); + const onceDescriptor = Object.getOwnPropertyDescriptor(page, "once"); + const removeAllDescriptor = Object.getOwnPropertyDescriptor(page, "removeAllListeners"); + + Object.defineProperties(page, { + on: { + configurable: true, + value: (type: unknown, handler: unknown): Page => { + Reflect.apply(on, page, [type, handler]); + if (type === "request") requestHandlers.push(handler); + return page; + }, + }, + once: { + configurable: true, + value: (type: unknown, handler: unknown): Page => { + if (type !== "request" || typeof handler !== "function") { + Reflect.apply(once, page, [type, handler]); + return page; + } + const wrapper = (event: unknown): void => { + const index = requestHandlers.lastIndexOf(wrapper); + if (index >= 0) requestHandlers.splice(index, 1); + Reflect.apply(off, page, ["request", wrapper]); + Reflect.apply(handler, page, [event]); + }; + requestHandlers.push(wrapper); + Reflect.apply(on, page, [type, wrapper]); + return page; + }, + }, + off: { + configurable: true, + value: (type: unknown, handler?: unknown): Page => { + Reflect.apply(off, page, [type, handler]); + if (type === "request") { + if (handler === undefined) requestHandlers.length = 0; + else { + const index = requestHandlers.lastIndexOf(handler); + if (index >= 0) requestHandlers.splice(index, 1); + } + } + return page; + }, + }, + removeAllListeners: { + configurable: true, + value: (type?: unknown): Page => { + Reflect.apply(removeAllListeners, page, [type]); + if (type === undefined || type === "request") requestHandlers.length = 0; + return page; + }, + }, + }); + + return { + page, + async cleanup() { + if (onDescriptor) Object.defineProperty(page, "on", onDescriptor); + else Reflect.deleteProperty(page, "on"); + if (offDescriptor) Object.defineProperty(page, "off", offDescriptor); + else Reflect.deleteProperty(page, "off"); + if (onceDescriptor) Object.defineProperty(page, "once", onceDescriptor); + else Reflect.deleteProperty(page, "once"); + if (removeAllDescriptor) Object.defineProperty(page, "removeAllListeners", removeAllDescriptor); + else Reflect.deleteProperty(page, "removeAllListeners"); + for (const handler of requestHandlers) Reflect.apply(off, page, ["request", handler]); + requestHandlers.length = 0; + try { + await withTimeout( + page.setRequestInterception(false), + REQUEST_INTERCEPTION_CLEANUP_TIMEOUT_MS, + "Timed out clearing browser request interception", + ); + } catch (error) { + throw new RequestInterceptionCleanupError( + "Failed to clear browser request interception after browser.run", + { + error: error instanceof Error ? error.message : String(error), + }, + ); + } + }, + }; +} + function errorPayload(error: unknown): RunErrorPayload { + const recoverTab = error instanceof RequestInterceptionCleanupError || undefined; if (error instanceof ToolAbortError) { return { name: error.name, message: error.message, stack: error.stack, isToolError: false, isAbort: true }; } if (error instanceof ToolError) { - return { name: error.name, message: error.message, stack: error.stack, isToolError: true, isAbort: false }; + return { + name: error.name, + message: error.message, + stack: error.stack, + isToolError: true, + isAbort: false, + recoverTab, + }; } if (error instanceof Error) { return { name: error.name, message: error.message, stack: error.stack, isToolError: false, isAbort: false }; @@ -720,6 +855,7 @@ export class WorkerCore { await session.send("Page.enable").catch(() => undefined); await session.send("Page.handleJavaScriptDialog", { accept: false }).catch(() => undefined); await session.send("Page.stopLoading").catch(() => undefined); + await session.send("Fetch.disable").catch(() => undefined); } catch (error) { this.#log("debug", "Recovery CDP session failed; proceeding with attach", { error: error instanceof Error ? error.message : String(error), @@ -816,17 +952,21 @@ export class WorkerCore { opCounter: 0, }; this.#active = active; + let completed = false; + let returnValue: unknown; + let failure: { error: unknown } | undefined; + let runPage: RunPageScope | undefined; try { throwIfAborted(signal); - const page = this.#requirePage(); + runPage = createRunPageScope(this.#requirePage()); const browser = this.#requireBrowser(); const tabApi = this.#createTabApi(msg.name, msg.timeoutMs, signal, msg.session, output, screenshots, active); const runtime = this.#ensureRuntime(msg.session); runtime.setCwd(msg.session.cwd); runtime.setRunScope({ - page, - browser, - tab: tabApi, + page: bindBrowserRunFacade(runPage.page, signal), + browser: bindBrowserRunFacade(browser, signal), + tab: bindBrowserRunFacade(tabApi, signal), assert: (cond: unknown, text?: string): void => { if (!cond) throw new ToolError(text ?? "Assertion failed"); }, @@ -879,25 +1019,37 @@ export class WorkerCore { try { const hooks = this.#hooksForActiveRun(); if (!hooks) throw new ToolError("Browser runtime started without an active run"); - const returnValue = await Promise.race([ + returnValue = await Promise.race([ runtime.run(msg.code, `browser-run-${msg.id}.js`, hooks, { runId: msg.id, cwd: msg.session.cwd }), cancelRejection, ]); - await this.#postReadyInfo(); - this.#transport.send({ - type: "result", - id: msg.id, - ok: true, - payload: { displays: output.finish(), returnValue: cloneSafe(returnValue), screenshots }, - }); + completed = true; } finally { signal.removeEventListener("abort", onCancel); } } catch (error) { - this.#transport.send({ type: "result", id: msg.id, ok: false, error: errorPayload(error) }); + failure = { error }; } finally { - if (this.#active?.id === msg.id) this.#active = null; runAc.abort(postmortem.markExpectedCleanupError(new ToolAbortError("Browser run ended"))); + try { + await runPage?.cleanup(); + } catch (error) { + failure = { error }; + } + if (this.#active?.id === msg.id) this.#active = null; + } + if (failure) { + this.#transport.send({ type: "result", id: msg.id, ok: false, error: errorPayload(failure.error) }); + return; + } + if (completed) { + await this.#postReadyInfo(); + this.#transport.send({ + type: "result", + id: msg.id, + ok: true, + payload: { displays: output.finish(), returnValue: cloneSafe(returnValue), screenshots }, + }); } } @@ -1212,11 +1364,22 @@ export class WorkerCore { press: (key, opts) => op(`tab.press(${JSON.stringify(key)})`, actionOpMs, async sig => { const selector = opts?.selector; - if (selector) await untilAborted(sig, () => page.focus(normalizeSelector(selector))); + if (selector) { + if (parseAriaRefSelector(selector) !== null) { + const handle = await this.#resolveAriaRef(selector); + try { + await untilAborted(sig, () => handle.focus()); + } finally { + await handle.dispose().catch(() => undefined); + } + } else await untilAborted(sig, () => page.focus(normalizeSelector(selector))); + } await untilAborted(sig, () => page.keyboard.press(key)); }), scroll: (deltaX, deltaY) => - op("tab.scroll()", actionOpMs, sig => untilAborted(sig, () => page.mouse.wheel({ deltaX, deltaY }))), + op("tab.scroll()", actionOpMs, sig => + untilAborted(sig, () => dispatchScroll(() => page.mouse.wheel({ deltaX, deltaY }))), + ), drag: (from, to) => op("tab.drag()", actionOpMs, sig => this.#drag(from, to, sig)), waitFor: (selector, opts) => { const w = waitMs(opts?.timeout); @@ -1387,9 +1550,10 @@ export class WorkerCore { const captureMime = `image/${captureType}` as const; let buffer: Buffer; if (opts.selector) { - const handle = (await untilAborted(signal, () => - page.$(normalizeSelector(opts.selector!)), - )) as ElementHandle | null; + const handle = + parseAriaRefSelector(opts.selector) !== null + ? await this.#resolveAriaRef(opts.selector) + : asElementHandle(await untilAborted(signal, () => page.$(normalizeSelector(opts.selector!)))); if (!handle) throw new ToolError("Screenshot selector did not resolve to an element"); try { // Bring the element into view with a single instant scroll instead of puppeteer's @@ -1462,9 +1626,10 @@ export class WorkerCore { role: "from" | "to", ): Promise<{ x: number; y: number; handle?: ElementHandle }> => { if (typeof target === "string") { - const handle = (await untilAborted(signal, () => - page.$(normalizeSelector(target)), - )) as ElementHandle | null; + const handle = + parseAriaRefSelector(target) !== null + ? await this.#resolveAriaRef(target) + : asElementHandle(await untilAborted(signal, () => page.$(normalizeSelector(target)))); if (!handle) throw new ToolError(`Drag ${role} selector did not resolve: ${target}`); const box = (await untilAborted(signal, () => handle.boundingBox())) as { x: number; @@ -1505,10 +1670,7 @@ export class WorkerCore { } async #select(selector: string, values: string[], timeoutMs: number, signal: AbortSignal): Promise { - const page = this.#requirePage(); - const handle = (await untilAborted(signal, () => - page.locator(normalizeSelector(selector)).setTimeout(timeoutMs).waitHandle({ signal }), - )) as ElementHandle; + const handle = await this.#resolveActionHandle(selector, timeoutMs, signal); try { return (await untilAborted(signal, () => handle.evaluate((el, vals) => { @@ -1527,10 +1689,17 @@ export class WorkerCore { globalThis as unknown as { Event: new (type: string, init?: { bubbles: boolean }) => unknown } ).Event; const wanted = new Set(vals as string[]); - const selected: string[] = []; + // Assign the full selection first, then read back: on a single + //